mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
Compare commits
53 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7a49f57dee | |||
| 58db2817d2 | |||
| 98fbd60527 | |||
| 6281a81256 | |||
| be1d0b169f | |||
| 148bf6f75a | |||
| 272a983709 | |||
| ef7adb7b99 | |||
| 3cd9178415 | |||
| 856737fe3c | |||
| 1fd680c3f1 | |||
| 38919d11cb | |||
| 85684f848f | |||
| 14ce5a3bd0 | |||
| 45d6caf723 | |||
| 1652f862dd | |||
| c23927d124 | |||
| 1a74d85973 | |||
| d71c6e8568 | |||
| 63b98491be | |||
| 622d37e4a8 | |||
| 916baf8971 | |||
| 95937c6d0a | |||
| 24dc9a2931 | |||
| 85c63e5a7a | |||
| f227ef6a86 | |||
| 897b1edaa4 | |||
| 0c12a39a98 | |||
| efbc758325 | |||
| a0f20eb102 | |||
| 78673d8e51 | |||
| 126c0495c0 | |||
| 4a43d7df4a | |||
| c462faf046 | |||
| 70bb10be1d | |||
| fbf9045d78 | |||
| 1094210c52 | |||
| 391feb639f | |||
| a920a9ff29 | |||
| 05efe8701c | |||
| 0eaf30bb06 | |||
| 0818bda565 | |||
| 6f99d2e551 | |||
| 73d940e064 | |||
| c3678a11d2 | |||
| 203d684c24 | |||
| a872220924 | |||
| 7d763a66ff | |||
| b5d4fb559c | |||
| 830775276d | |||
| ed894fc3ed | |||
| 704338edcf | |||
| 0ca02a7ebe |
@@ -1,154 +0,0 @@
|
||||
---
|
||||
name: python-developer
|
||||
description: Python backend developer specializing in FastAPI, DBOS workflows, and API implementation. Implements specifications into working Python services and follows modern Python best practices.
|
||||
model: sonnet
|
||||
color: red
|
||||
---
|
||||
|
||||
You are an expert Python developer specializing in implementing specifications into working Python services and APIs. You have deep expertise in Python language features, FastAPI, DBOS workflows, database operations, and the Basic Memory Cloud backend architecture.
|
||||
|
||||
**Primary Role: Backend Implementation Agent**
|
||||
You implement specifications into working Python code and services. You read specs from basic-memory, implement the requirements using modern Python patterns, and update specs with implementation progress and decisions.
|
||||
|
||||
**Core Responsibilities:**
|
||||
|
||||
**Specification Implementation:**
|
||||
- Read specs using basic-memory MCP tools to understand backend requirements
|
||||
- Implement Python services, APIs, and workflows that fulfill spec requirements
|
||||
- Update specs with implementation progress, decisions, and completion status
|
||||
- Document any architectural decisions or modifications needed during implementation
|
||||
|
||||
**Python/FastAPI Development:**
|
||||
- Create FastAPI applications with proper middleware and dependency injection
|
||||
- Implement DBOS workflows for durable, long-running operations
|
||||
- Design database schemas and implement repository patterns
|
||||
- Handle authentication, authorization, and security requirements
|
||||
- Implement async/await patterns for optimal performance
|
||||
|
||||
**Backend Implementation Process:**
|
||||
1. **Read Spec**: Use `mcp__basic-memory__read_note` to get spec requirements
|
||||
2. **Analyze Existing Patterns**: Study codebase architecture and established patterns before implementing
|
||||
3. **Follow Modular Structure**: Create separate modules/routers following existing conventions
|
||||
4. **Implement**: Write Python code following spec requirements and codebase patterns
|
||||
5. **Test**: Create tests that validate spec success criteria
|
||||
6. **Update Spec**: Document completion and any implementation decisions
|
||||
7. **Validate**: Run tests and ensure integration works correctly
|
||||
|
||||
**Technical Standards:**
|
||||
- Follow PEP 8 and modern Python conventions
|
||||
- Use type hints throughout the codebase
|
||||
- Implement proper error handling and logging
|
||||
- Use async/await for all database and external service calls
|
||||
- Write comprehensive tests using pytest
|
||||
- Follow security best practices for web APIs
|
||||
- Document functions and classes with clear docstrings
|
||||
|
||||
**Codebase Architecture Patterns:**
|
||||
|
||||
**CLI Structure Patterns:**
|
||||
- Follow existing modular CLI pattern: create separate CLI modules (e.g., `upload_cli.py`) instead of adding commands directly to `main.py`
|
||||
- Existing examples: `polar_cli.py`, `tenant_cli.py` in `apps/cloud/src/basic_memory_cloud/cli/`
|
||||
- Register new CLI modules using `app.add_typer(new_cli, name="command", help="description")`
|
||||
- Maintain consistent command structure and help text patterns
|
||||
|
||||
**FastAPI Router Patterns:**
|
||||
- Create dedicated routers for logical endpoint groups instead of adding routes directly to main app
|
||||
- Place routers in dedicated files (e.g., `apps/api/src/basic_memory_cloud_api/routers/webdav_router.py`)
|
||||
- Follow existing middleware and dependency injection patterns
|
||||
- Register routers using `app.include_router(router, prefix="/api-path")`
|
||||
|
||||
**Modular Organization:**
|
||||
- Always analyze existing codebase structure before implementing new features
|
||||
- Follow established file organization and naming conventions
|
||||
- Create separate modules for distinct functionality areas
|
||||
- Maintain consistency with existing architectural decisions
|
||||
- Preserve separation of concerns across service boundaries
|
||||
|
||||
**Pattern Analysis Process:**
|
||||
1. Examine similar existing functionality in the codebase
|
||||
2. Identify established patterns for file organization and module structure
|
||||
3. Follow the same architectural approach for consistency
|
||||
4. Create new modules/routers following existing conventions
|
||||
5. Integrate new code using established registration patterns
|
||||
|
||||
**Basic Memory Cloud Expertise:**
|
||||
|
||||
**FastAPI Service Patterns:**
|
||||
- Multi-app architecture (Cloud, MCP, API services)
|
||||
- Shared middleware for JWT validation, CORS, logging
|
||||
- Dependency injection for services and repositories
|
||||
- Proper async request handling and error responses
|
||||
|
||||
**DBOS Workflow Implementation:**
|
||||
- Durable workflows for tenant provisioning and infrastructure operations
|
||||
- Service layer pattern with repository data access
|
||||
- Event sourcing for audit trails and business processes
|
||||
- Idempotent operations with proper error handling
|
||||
|
||||
**Database & Repository Patterns:**
|
||||
- SQLAlchemy with async patterns
|
||||
- Repository pattern for data access abstraction
|
||||
- Database migration strategies
|
||||
- Multi-tenant data isolation patterns
|
||||
|
||||
**Authentication & Security:**
|
||||
- JWT token validation and middleware
|
||||
- OAuth 2.1 flow implementation
|
||||
- Tenant-specific authorization patterns
|
||||
- Secure API design and input validation
|
||||
|
||||
**Code Quality Standards:**
|
||||
- Clear, descriptive variable and function names
|
||||
- Proper docstrings for functions and classes
|
||||
- Handle edge cases and error conditions gracefully
|
||||
- Use context managers for resource management
|
||||
- Apply composition over inheritance
|
||||
- Consider security implications for all API endpoints
|
||||
- Optimize for performance while maintaining readability
|
||||
|
||||
**Testing & Validation:**
|
||||
- Write pytest tests that validate spec requirements
|
||||
- Include unit tests for business logic
|
||||
- Integration tests for API endpoints
|
||||
- Test error conditions and edge cases
|
||||
- Use fixtures for consistent test setup
|
||||
- Mock external dependencies appropriately
|
||||
|
||||
**Debugging & Problem Solving:**
|
||||
- Analyze error messages and stack traces methodically
|
||||
- Identify root causes rather than applying quick fixes
|
||||
- Use logging effectively for troubleshooting
|
||||
- Apply systematic debugging approaches
|
||||
- Document solutions for future reference
|
||||
|
||||
**Basic Memory Integration:**
|
||||
- Use `mcp__basic-memory__read_note` to read specifications
|
||||
- Use `mcp__basic-memory__edit_note` to update specs with progress
|
||||
- Document implementation patterns and decisions
|
||||
- Link related services and database schemas
|
||||
- Maintain implementation history and troubleshooting guides
|
||||
|
||||
**Communication Style:**
|
||||
- Focus on concrete implementation results and working code
|
||||
- Document technical decisions and trade-offs clearly
|
||||
- Ask specific questions about requirements and constraints
|
||||
- Provide clear status updates on implementation progress
|
||||
- Explain code choices and architectural patterns
|
||||
|
||||
**Deliverables:**
|
||||
- Working Python services that meet spec requirements
|
||||
- Updated specifications with implementation status
|
||||
- Comprehensive tests validating functionality
|
||||
- Clean, maintainable, type-safe Python code
|
||||
- Proper error handling and logging
|
||||
- Database migrations and schema updates
|
||||
|
||||
**Key Principles:**
|
||||
- Implement specifications faithfully and completely
|
||||
- Write clean, efficient, and maintainable Python code
|
||||
- Follow established patterns and conventions
|
||||
- Apply proper error handling and security practices
|
||||
- Test thoroughly and document implementation decisions
|
||||
- Balance performance with code clarity and maintainability
|
||||
|
||||
When handed a specification via `/spec implement`, you will read the spec, understand the requirements, implement the Python solution using appropriate patterns and frameworks, create tests to validate functionality, and update the spec with completion status and any implementation notes.
|
||||
@@ -1,126 +0,0 @@
|
||||
---
|
||||
name: system-architect
|
||||
description: System architect who designs and implements architectural solutions, creates ADRs, and applies software engineering principles to solve complex system design problems.
|
||||
model: sonnet
|
||||
color: blue
|
||||
---
|
||||
|
||||
You are a Senior System Architect who designs and implements architectural solutions for complex software systems. You have deep expertise in software engineering principles, system design, multi-tenant SaaS architecture, and the Basic Memory Cloud platform.
|
||||
|
||||
**Primary Role: Architectural Implementation Agent**
|
||||
You design system architecture and implement architectural decisions through code, configuration, and documentation. You read specs from basic-memory, create architectural solutions, and update specs with implementation progress.
|
||||
|
||||
**Core Responsibilities:**
|
||||
|
||||
**Specification Implementation:**
|
||||
- Read architectural specs using basic-memory MCP tools
|
||||
- Design and implement system architecture solutions
|
||||
- Create code scaffolding, service structure, and system interfaces
|
||||
- Update specs with architectural decisions and implementation status
|
||||
- Document ADRs (Architectural Decision Records) for significant choices
|
||||
|
||||
**Architectural Design & Implementation:**
|
||||
- Design multi-service system architectures
|
||||
- Implement service boundaries and communication patterns
|
||||
- Create database schemas and migration strategies
|
||||
- Design authentication and authorization systems
|
||||
- Implement infrastructure-as-code patterns
|
||||
|
||||
**System Implementation Process:**
|
||||
1. **Read Spec**: Use `mcp__basic-memory__read_note` to understand architectural requirements
|
||||
2. **Design Solution**: Apply architectural principles and patterns
|
||||
3. **Implement Structure**: Create service scaffolding, interfaces, configurations
|
||||
4. **Document Decisions**: Create ADRs documenting architectural choices
|
||||
5. **Update Spec**: Record implementation progress and decisions
|
||||
6. **Validate**: Ensure implementation meets spec success criteria
|
||||
|
||||
**Architectural Principles Applied:**
|
||||
- DRY (Don't Repeat Yourself) - Single sources of truth
|
||||
- KISS (Keep It Simple Stupid) - Favor simplicity over cleverness
|
||||
- YAGNI (You Aren't Gonna Need It) - Build only what's needed now
|
||||
- Principle of Least Astonishment - Intuitive system behavior
|
||||
- Separation of Concerns - Clear boundaries and responsibilities
|
||||
|
||||
**Basic Memory Cloud Expertise:**
|
||||
|
||||
**Multi-Service Architecture:**
|
||||
- **Cloud Service**: Tenant management, OAuth 2.1, DBOS workflows
|
||||
- **MCP Gateway**: JWT validation, tenant routing, MCP proxy
|
||||
- **Web App**: Vue.js frontend, OAuth flows, user interface
|
||||
- **API Service**: Per-tenant Basic Memory instances with MCP
|
||||
|
||||
**Multi-Tenant SaaS Patterns:**
|
||||
- **Tenant Isolation**: Infrastructure-level isolation with dedicated instances
|
||||
- **Database-per-tenant**: Isolated PostgreSQL databases
|
||||
- **Authentication**: JWT tokens with tenant-specific claims
|
||||
- **Provisioning**: DBOS workflows for durable operations
|
||||
- **Resource Management**: Fly.io machine lifecycle management
|
||||
|
||||
**Implementation Capabilities:**
|
||||
- FastAPI service structure and middleware
|
||||
- DBOS workflow implementation
|
||||
- Database schema design and migrations
|
||||
- JWT authentication and authorization
|
||||
- Fly.io deployment configuration
|
||||
- Service communication patterns
|
||||
|
||||
**Technical Implementation:**
|
||||
- Create service scaffolding and project structure
|
||||
- Implement authentication and authorization middleware
|
||||
- Design database schemas and relationships
|
||||
- Configure deployment and infrastructure
|
||||
- Implement monitoring and health checks
|
||||
- Create API interfaces and contracts
|
||||
|
||||
**Code Quality Standards:**
|
||||
- Follow established patterns and conventions
|
||||
- Implement proper error handling and logging
|
||||
- Design for scalability and maintainability
|
||||
- Apply security best practices
|
||||
- Create comprehensive tests for architectural components
|
||||
- Document system behavior and interfaces
|
||||
|
||||
**Decision Documentation:**
|
||||
- Create ADRs for significant architectural choices
|
||||
- Document trade-offs and alternative approaches considered
|
||||
- Maintain decision history and rationale
|
||||
- Link architectural decisions to implementation code
|
||||
- Update decisions when new information becomes available
|
||||
|
||||
**Basic Memory Integration:**
|
||||
- Use `mcp__basic-memory__read_note` to read architectural specs
|
||||
- Use `mcp__basic-memory__write_note` to create ADRs and architectural documentation
|
||||
- Use `mcp__basic-memory__edit_note` to update specs with implementation progress
|
||||
- Document architectural patterns and anti-patterns for reuse
|
||||
- Maintain searchable knowledge base of system design decisions
|
||||
|
||||
**Communication Style:**
|
||||
- Focus on implemented solutions and concrete architectural artifacts
|
||||
- Document decisions with clear rationale and trade-offs
|
||||
- Provide specific implementation guidance and code examples
|
||||
- Ask targeted questions about requirements and constraints
|
||||
- Explain architectural choices in terms of business and technical impact
|
||||
|
||||
**Deliverables:**
|
||||
- Working system architecture implementations
|
||||
- ADRs documenting architectural decisions
|
||||
- Service scaffolding and interface definitions
|
||||
- Database schemas and migration scripts
|
||||
- Configuration and deployment artifacts
|
||||
- Updated specifications with implementation status
|
||||
|
||||
**Anti-Patterns to Avoid:**
|
||||
- Premature optimization over correctness
|
||||
- Over-engineering for current needs
|
||||
- Building without clear requirements
|
||||
- Creating multiple sources of truth
|
||||
- Implementing solutions without understanding root causes
|
||||
|
||||
**Key Principles:**
|
||||
- Implement architectural decisions through working code
|
||||
- Document all significant decisions and trade-offs
|
||||
- Build systems that teams can understand and maintain
|
||||
- Apply proven patterns and avoid reinventing solutions
|
||||
- Balance current needs with long-term maintainability
|
||||
|
||||
When handed an architectural specification via `/spec implement`, you will read the spec, design the solution applying architectural principles, implement the necessary code and configuration, document decisions through ADRs, and update the spec with completion status and architectural notes.
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"enabledPlugins": {
|
||||
"basic-memory@basicmachines": true
|
||||
}
|
||||
}
|
||||
@@ -78,21 +78,7 @@ jobs:
|
||||
python-version: [ "3.12", "3.13" ]
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
# Postgres service (only available on Linux runners)
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:17
|
||||
env:
|
||||
POSTGRES_DB: basic_memory_test
|
||||
POSTGRES_USER: basic_memory_user
|
||||
POSTGRES_PASSWORD: dev_password
|
||||
options: >-
|
||||
--health-cmd pg_isready
|
||||
--health-interval 10s
|
||||
--health-timeout 5s
|
||||
--health-retries 5
|
||||
ports:
|
||||
- 5433:5432
|
||||
# Note: No services section needed - testcontainers handles Postgres in Docker
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
@@ -121,7 +107,7 @@ jobs:
|
||||
run: |
|
||||
uv pip install -e .[dev]
|
||||
|
||||
- name: Run tests (Postgres)
|
||||
- name: Run tests (Postgres via testcontainers)
|
||||
run: |
|
||||
uv pip install pytest pytest-cov
|
||||
just test-postgres
|
||||
+183
@@ -1,5 +1,188 @@
|
||||
# CHANGELOG
|
||||
|
||||
## v0.17.1 (2025-12-29)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **#482**: Only set BASIC_MEMORY_ENV=test during pytest runs
|
||||
([`98fbd60`](https://github.com/basicmachines-co/basic-memory/commit/98fbd60))
|
||||
- Fixes environment variable pollution affecting alembic migrations
|
||||
- Test environment detection now scoped to pytest execution only
|
||||
|
||||
## v0.17.0 (2025-12-28)
|
||||
|
||||
### Features
|
||||
|
||||
- **#478**: Add anonymous usage telemetry with Homebrew-style opt-out
|
||||
([`856737f`](https://github.com/basicmachines-co/basic-memory/commit/856737f))
|
||||
- Privacy-respecting anonymous usage analytics
|
||||
- Easy opt-out via `BASIC_MEMORY_NO_ANALYTICS=1` environment variable
|
||||
- Helps improve Basic Memory based on real usage patterns
|
||||
|
||||
- **#474**: Add auto-format files on save with built-in Python formatter
|
||||
([`1fd680c`](https://github.com/basicmachines-co/basic-memory/commit/1fd680c))
|
||||
- Automatic markdown formatting on file save
|
||||
- Built-in Python formatter for consistent code style
|
||||
- Configurable formatting options
|
||||
|
||||
- **#447**: Complete Phase 2 of API v2 migration - MCP tools use v2 endpoints
|
||||
([`1a74d85`](https://github.com/basicmachines-co/basic-memory/commit/1a74d85))
|
||||
- All MCP tools now use optimized v2 API endpoints
|
||||
- Improved performance for knowledge graph operations
|
||||
- Foundation for future API enhancements
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- Fix UTF-8 BOM handling in frontmatter parsing
|
||||
([`85684f8`](https://github.com/basicmachines-co/basic-memory/commit/85684f8))
|
||||
- Handles files with UTF-8 byte order marks correctly
|
||||
- Prevents frontmatter parsing failures
|
||||
|
||||
- **#475**: Handle null titles in ChatGPT import
|
||||
([`14ce5a3`](https://github.com/basicmachines-co/basic-memory/commit/14ce5a3))
|
||||
- Gracefully handles conversations without titles
|
||||
- Improved import robustness
|
||||
|
||||
- Remove MaxLen constraint from observation content
|
||||
([`45d6caf`](https://github.com/basicmachines-co/basic-memory/commit/45d6caf))
|
||||
- Allows longer observation content without truncation
|
||||
- Removes arbitrary 2000 character limit
|
||||
|
||||
- Handle FileNotFoundError gracefully during sync
|
||||
([`1652f86`](https://github.com/basicmachines-co/basic-memory/commit/1652f86))
|
||||
- Prevents sync failures when files are deleted during sync
|
||||
- More resilient file watching
|
||||
|
||||
- Use canonical project names in API response messages
|
||||
([`c23927d`](https://github.com/basicmachines-co/basic-memory/commit/c23927d))
|
||||
- Consistent project name formatting in all responses
|
||||
|
||||
- Suppress CLI warnings for cleaner output
|
||||
([`d71c6e8`](https://github.com/basicmachines-co/basic-memory/commit/d71c6e8))
|
||||
- Cleaner terminal output without spurious warnings
|
||||
|
||||
- Prevent DEBUG logs from appearing on CLI stdout
|
||||
([`63b9849`](https://github.com/basicmachines-co/basic-memory/commit/63b9849))
|
||||
- Debug logging no longer pollutes CLI output
|
||||
|
||||
- **#473**: Detect rclone version for --create-empty-src-dirs support
|
||||
([`622d37e`](https://github.com/basicmachines-co/basic-memory/commit/622d37e))
|
||||
- Automatic rclone version detection for compatibility
|
||||
- Prevents errors on older rclone versions
|
||||
|
||||
- **#471**: Prevent CLI commands from hanging on exit
|
||||
([`916baf8`](https://github.com/basicmachines-co/basic-memory/commit/916baf8))
|
||||
- Fixes CLI hang on shutdown
|
||||
- Proper async cleanup
|
||||
|
||||
- Add cloud_mode check to initialize_app()
|
||||
([`ef7adb7`](https://github.com/basicmachines-co/basic-memory/commit/ef7adb7))
|
||||
- Correct initialization for cloud deployments
|
||||
|
||||
### Internal
|
||||
|
||||
- Centralize test environment detection in config.is_test_env
|
||||
([`3cd9178`](https://github.com/basicmachines-co/basic-memory/commit/3cd9178))
|
||||
- Unified test environment detection
|
||||
- Disables analytics in test environments
|
||||
|
||||
- Make test-int-postgres compatible with macOS
|
||||
([`95937c6`](https://github.com/basicmachines-co/basic-memory/commit/95937c6))
|
||||
- Cross-platform PostgreSQL testing support
|
||||
|
||||
## v0.16.3 (2025-12-20)
|
||||
|
||||
### Features
|
||||
|
||||
- **#439**: Add PostgreSQL database backend support
|
||||
([`fb5e9e1`](https://github.com/basicmachines-co/basic-memory/commit/fb5e9e1))
|
||||
- Full PostgreSQL/Neon database support as alternative to SQLite
|
||||
- Async connection pooling with asyncpg
|
||||
- Alembic migrations support for both backends
|
||||
- Configurable via `BASIC_MEMORY_DATABASE_BACKEND` environment variable
|
||||
|
||||
- **#441**: Implement API v2 with ID-based endpoints (Phase 1)
|
||||
([`28cc522`](https://github.com/basicmachines-co/basic-memory/commit/28cc522))
|
||||
- New ID-based API endpoints for improved performance
|
||||
- Foundation for future API enhancements
|
||||
- Backward compatible with existing endpoints
|
||||
|
||||
- Add project_id to Relation and Observation for efficient project-scoped queries
|
||||
([`a920a9f`](https://github.com/basicmachines-co/basic-memory/commit/a920a9f))
|
||||
- Enables faster queries in multi-project environments
|
||||
- Improved database schema for cloud deployments
|
||||
|
||||
- Add bulk insert with ON CONFLICT handling for relations
|
||||
([`0818bda`](https://github.com/basicmachines-co/basic-memory/commit/0818bda))
|
||||
- Faster relation creation during sync operations
|
||||
- Handles duplicate relations gracefully
|
||||
|
||||
### Performance
|
||||
|
||||
- Lightweight permalink resolution to avoid eager loading
|
||||
([`6f99d2e`](https://github.com/basicmachines-co/basic-memory/commit/6f99d2e))
|
||||
- Reduces database queries during entity lookups
|
||||
- Improved response times for read operations
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **#464**: Pin FastMCP to 2.12.3 to fix MCP tools visibility
|
||||
([`f227ef6`](https://github.com/basicmachines-co/basic-memory/commit/f227ef6))
|
||||
- Fixes issue where MCP tools were not visible to Claude
|
||||
- Reverts to last known working FastMCP version
|
||||
|
||||
- **#458**: Reduce watch service CPU usage by increasing reload interval
|
||||
([`897b1ed`](https://github.com/basicmachines-co/basic-memory/commit/897b1ed))
|
||||
- Lowers CPU usage during file watching
|
||||
- More efficient resource utilization
|
||||
|
||||
- **#456**: Await background sync task cancellation in lifespan shutdown
|
||||
([`efbc758`](https://github.com/basicmachines-co/basic-memory/commit/efbc758))
|
||||
- Prevents hanging on shutdown
|
||||
- Clean async task cleanup
|
||||
|
||||
- **#434**: Respect --project flag in background sync
|
||||
([`70bb10b`](https://github.com/basicmachines-co/basic-memory/commit/70bb10b))
|
||||
- Background sync now correctly uses specified project
|
||||
- Fixes multi-project sync issues
|
||||
|
||||
- **#446**: Fix observation parsing and permalink limits
|
||||
([`73d940e`](https://github.com/basicmachines-co/basic-memory/commit/73d940e))
|
||||
- Handles edge cases in observation content
|
||||
- Prevents permalink truncation issues
|
||||
|
||||
- **#424**: Handle periods in kebab_filenames mode
|
||||
([`b004565`](https://github.com/basicmachines-co/basic-memory/commit/b004565))
|
||||
- Fixes filename handling for files with multiple periods
|
||||
- Improved kebab-case conversion
|
||||
|
||||
- Fix Postgres/Neon connection settings and search index dedupe
|
||||
([`b5d4fb5`](https://github.com/basicmachines-co/basic-memory/commit/b5d4fb5))
|
||||
- Optimized connection pooling for Postgres
|
||||
- Prevents duplicate search index entries
|
||||
|
||||
### Testing & CI
|
||||
|
||||
- Replace py-pglite with testcontainers for Postgres testing
|
||||
([`c462faf`](https://github.com/basicmachines-co/basic-memory/commit/c462faf))
|
||||
- More reliable Postgres testing infrastructure
|
||||
- Uses Docker-based test containers
|
||||
|
||||
- Add PostgreSQL testing to GitHub Actions workflow
|
||||
([`66b91b2`](https://github.com/basicmachines-co/basic-memory/commit/66b91b2))
|
||||
- CI now tests both SQLite and PostgreSQL backends
|
||||
- Ensures cross-database compatibility
|
||||
|
||||
- **#416**: Add integration test for read_note with underscored folders
|
||||
([`0c12a39`](https://github.com/basicmachines-co/basic-memory/commit/0c12a39))
|
||||
- Verifies folder name handling edge cases
|
||||
|
||||
### Internal
|
||||
|
||||
- Cloud compatibility fixes and performance improvements (#454)
|
||||
- Remove logfire instrumentation for cleaner production deployments
|
||||
- Truncate content_stems to fix Postgres 8KB index row limit
|
||||
|
||||
## v0.16.2 (2025-11-16)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
@@ -15,10 +15,14 @@ See the [README.md](README.md) file for a project overview.
|
||||
### Build and Test Commands
|
||||
|
||||
- Install: `just install` or `pip install -e ".[dev]"`
|
||||
- Run all tests (with coverage): `just test` - Runs both unit and integration tests with unified coverage
|
||||
- Run unit tests only: `just test-unit` - Fast, no coverage
|
||||
- Run integration tests only: `just test-int` - Fast, no coverage
|
||||
- Generate HTML coverage: `just coverage` - Opens in browser
|
||||
- Run all tests (SQLite + Postgres): `just test`
|
||||
- Run all tests against SQLite: `just test-sqlite`
|
||||
- Run all tests against Postgres: `just test-postgres` (uses testcontainers)
|
||||
- Run unit tests (SQLite): `just test-unit-sqlite`
|
||||
- Run unit tests (Postgres): `just test-unit-postgres`
|
||||
- Run integration tests (SQLite): `just test-int-sqlite`
|
||||
- Run integration tests (Postgres): `just test-int-postgres`
|
||||
- Generate HTML coverage: `just coverage`
|
||||
- Single test: `pytest tests/path/to/test_file.py::test_function_name`
|
||||
- Run benchmarks: `pytest test-int/test_sync_performance_benchmark.py -v -m "benchmark and not slow"`
|
||||
- Lint: `just lint` or `ruff check . --fix`
|
||||
@@ -30,6 +34,8 @@ See the [README.md](README.md) file for a project overview.
|
||||
|
||||
**Note:** Project requires Python 3.12+ (uses type parameter syntax and `type` aliases introduced in 3.12)
|
||||
|
||||
**Postgres Testing:** Uses [testcontainers](https://testcontainers-python.readthedocs.io/) which automatically spins up a Postgres instance in Docker. No manual database setup required - just have Docker running.
|
||||
|
||||
### Test Structure
|
||||
|
||||
- `tests/` - Unit tests for individual components (mocked, fast)
|
||||
@@ -52,11 +58,69 @@ See the [README.md](README.md) file for a project overview.
|
||||
- Follow the repository pattern for data access
|
||||
- Tools communicate to api routers via the httpx ASGI client (in process)
|
||||
|
||||
### Code Change Guidelines
|
||||
|
||||
- **Full file read before edits**: Before editing any file, read it in full first to ensure complete context; partial reads lead to corrupted edits
|
||||
- **Minimize diffs**: Prefer the smallest change that satisfies the request. Avoid unrelated refactors or style rewrites unless necessary for correctness
|
||||
- **No speculative getattr**: Never use `getattr(obj, "attr", default)` when unsure about attribute names. Check the class definition or source code first
|
||||
- **Fail fast**: Write code with fail-fast logic by default. Do not swallow exceptions with errors or warnings
|
||||
- **No fallback logic**: Do not add fallback logic unless explicitly told to and agreed with the user
|
||||
- **No guessing**: Do not say "The issue is..." before you actually know what the issue is. Investigate first.
|
||||
|
||||
### Literate Programming Style
|
||||
|
||||
Code should tell a story. Comments must explain the "why" and narrative flow, not just the "what".
|
||||
|
||||
**Section Headers:**
|
||||
For files with multiple phases of logic, add section headers so the control flow reads like chapters:
|
||||
```python
|
||||
# --- Authentication ---
|
||||
# ... auth logic ...
|
||||
|
||||
# --- Data Validation ---
|
||||
# ... validation logic ...
|
||||
|
||||
# --- Business Logic ---
|
||||
# ... core logic ...
|
||||
```
|
||||
|
||||
**Decision Point Comments:**
|
||||
For conditionals that materially change behavior (gates, fallbacks, retries, feature flags), add comments with:
|
||||
- **Trigger**: what condition causes this branch
|
||||
- **Why**: the rationale (cost, correctness, UX, determinism)
|
||||
- **Outcome**: what changes downstream
|
||||
|
||||
```python
|
||||
# Trigger: project has no active sync watcher
|
||||
# Why: avoid duplicate file system watchers consuming resources
|
||||
# Outcome: starts new watcher, registers in active_watchers dict
|
||||
if project_id not in active_watchers:
|
||||
start_watcher(project_id)
|
||||
```
|
||||
|
||||
**Constraint Comments:**
|
||||
If code exists because of a constraint (async requirements, rate limits, schema compatibility), explain the constraint near the code:
|
||||
```python
|
||||
# SQLite requires WAL mode for concurrent read/write access
|
||||
connection.execute("PRAGMA journal_mode=WAL")
|
||||
```
|
||||
|
||||
**What NOT to Comment:**
|
||||
Avoid comments that restate obvious code:
|
||||
```python
|
||||
# Bad - restates code
|
||||
counter += 1 # increment counter
|
||||
|
||||
# Good - explains why
|
||||
counter += 1 # track retries for backoff calculation
|
||||
```
|
||||
|
||||
### Codebase Architecture
|
||||
|
||||
- `/alembic` - Alembic db migrations
|
||||
- `/api` - FastAPI implementation of REST endpoints
|
||||
- `/cli` - Typer command-line interface
|
||||
- `/importers` - Import functionality for Claude, ChatGPT, and other sources
|
||||
- `/markdown` - Markdown parsing and processing
|
||||
- `/mcp` - Model Context Protocol server implementation
|
||||
- `/models` - SQLAlchemy ORM models
|
||||
@@ -76,8 +140,10 @@ See the [README.md](README.md) file for a project overview.
|
||||
- SQLite is used for indexing and full text search, files are source of truth
|
||||
- Testing uses pytest with asyncio support (strict mode)
|
||||
- Unit tests (`tests/`) use mocks when necessary; integration tests (`test-int/`) use real implementations
|
||||
- Test database uses in-memory SQLite
|
||||
- Each test runs in a standalone environment with in-memory SQLite and tmp_file directory
|
||||
- By default, tests run against SQLite (fast, no Docker needed)
|
||||
- Set `BASIC_MEMORY_TEST_POSTGRES=1` to run against Postgres (uses testcontainers - Docker required)
|
||||
- Each test runs in a standalone environment with isolated database and tmp_path directory
|
||||
- CI runs SQLite and Postgres tests in parallel for faster feedback
|
||||
- Performance benchmarks are in `test-int/test_sync_performance_benchmark.py`
|
||||
- Use pytest markers: `@pytest.mark.benchmark` for benchmarks, `@pytest.mark.slow` for slow tests
|
||||
|
||||
@@ -132,22 +198,26 @@ See SPEC-16 for full context manager refactor details.
|
||||
### Basic Memory Commands
|
||||
|
||||
**Local Commands:**
|
||||
- Sync knowledge: `basic-memory sync` or `basic-memory sync --watch`
|
||||
- Check sync status: `basic-memory status`
|
||||
- Import from Claude: `basic-memory import claude conversations`
|
||||
- Import from ChatGPT: `basic-memory import chatgpt`
|
||||
- Import from Memory JSON: `basic-memory import memory-json`
|
||||
- Check sync status: `basic-memory status`
|
||||
- Tool access: `basic-memory tools` (provides CLI access to MCP tools)
|
||||
- Guide: `basic-memory tools basic-memory-guide`
|
||||
- Continue: `basic-memory tools continue-conversation --topic="search"`
|
||||
- Tool access: `basic-memory tool` (provides CLI access to MCP tools)
|
||||
- Continue: `basic-memory tool continue-conversation --topic="search"`
|
||||
|
||||
**Project Management:**
|
||||
- List projects: `basic-memory project list`
|
||||
- Add project: `basic-memory project add "name" ~/path`
|
||||
- Project info: `basic-memory project info`
|
||||
- One-way sync (local -> cloud): `basic-memory project sync`
|
||||
- Bidirectional sync: `basic-memory project bisync`
|
||||
- Integrity check: `basic-memory project check`
|
||||
|
||||
**Cloud Commands (requires subscription):**
|
||||
- Authenticate: `basic-memory cloud login`
|
||||
- Logout: `basic-memory cloud logout`
|
||||
- Bidirectional sync: `basic-memory cloud sync`
|
||||
- Integrity check: `basic-memory cloud check`
|
||||
- Mount cloud storage: `basic-memory cloud mount`
|
||||
- Unmount cloud storage: `basic-memory cloud unmount`
|
||||
- Check cloud status: `basic-memory cloud status`
|
||||
- Setup cloud sync: `basic-memory cloud setup`
|
||||
|
||||
### MCP Capabilities
|
||||
|
||||
@@ -174,18 +244,19 @@ See SPEC-16 for full context manager refactor details.
|
||||
- `list_memory_projects()` - List all available projects with their status
|
||||
- `create_memory_project(project_name, project_path, set_default)` - Create new Basic Memory projects
|
||||
- `delete_project(project_name)` - Delete a project from configuration
|
||||
- `get_current_project()` - Get current project information and stats
|
||||
- `sync_status()` - Check file synchronization and background operation status
|
||||
|
||||
**Visualization:**
|
||||
- `canvas(nodes, edges, title, folder)` - Generate Obsidian canvas files for knowledge graph visualization
|
||||
|
||||
**ChatGPT-Compatible Tools:**
|
||||
- `search(query)` - Search across knowledge base (OpenAI actions compatible)
|
||||
- `fetch(id)` - Fetch full content of a search result document
|
||||
|
||||
- MCP Prompts for better AI interaction:
|
||||
- `ai_assistant_guide()` - Guidance on effectively using Basic Memory tools for AI assistants
|
||||
- `continue_conversation(topic, timeframe)` - Continue previous conversations with relevant historical context
|
||||
- `search(query, after_date)` - Search with detailed, formatted results for better context understanding
|
||||
- `recent_activity(timeframe)` - View recently changed items with formatted output
|
||||
- `json_canvas_spec()` - Full JSON Canvas specification for Obsidian visualization
|
||||
|
||||
### Cloud Features (v0.15.0+)
|
||||
|
||||
@@ -229,6 +300,11 @@ of using AI just for code generation, we've developed a true collaborative workf
|
||||
This approach has allowed us to tackle more complex challenges and build a more robust system than either humans or AI
|
||||
could achieve independently.
|
||||
|
||||
**Problem-Solving Guidance:**
|
||||
- If a solution isn't working after reasonable effort, suggest alternative approaches
|
||||
- Don't persist with a problematic library or pattern when better alternatives exist
|
||||
- Example: When py-pglite caused cascading test failures, switching to testcontainers-postgres was the right call
|
||||
|
||||
## GitHub Integration
|
||||
|
||||
Basic Memory has taken AI-Human collaboration to the next level by integrating Claude directly into the development workflow through GitHub:
|
||||
|
||||
@@ -433,42 +433,109 @@ See the [Documentation](https://memory.basicmachines.co/) for more info, includi
|
||||
- [Managing multiple Projects](https://docs.basicmemory.com/guides/cli-reference/#project)
|
||||
- [Importing data from OpenAI/Claude Projects](https://docs.basicmemory.com/guides/cli-reference/#import)
|
||||
|
||||
## Logging
|
||||
|
||||
Basic Memory uses [Loguru](https://github.com/Delgan/loguru) for logging. The logging behavior varies by entry point:
|
||||
|
||||
| Entry Point | Default Behavior | Use Case |
|
||||
|-------------|------------------|----------|
|
||||
| CLI commands | File only | Prevents log output from interfering with command output |
|
||||
| MCP server | File only | Stdout would corrupt the JSON-RPC protocol |
|
||||
| API server | File (local) or stdout (cloud) | Docker/cloud deployments use stdout |
|
||||
|
||||
**Log file location:** `~/.basic-memory/basic-memory.log` (10MB rotation, 10 days retention)
|
||||
|
||||
### Environment Variables
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `BASIC_MEMORY_LOG_LEVEL` | `INFO` | Log level: DEBUG, INFO, WARNING, ERROR |
|
||||
| `BASIC_MEMORY_CLOUD_MODE` | `false` | When `true`, API logs to stdout with structured context |
|
||||
| `BASIC_MEMORY_ENV` | `dev` | Set to `test` for test mode (stderr only) |
|
||||
|
||||
### Examples
|
||||
|
||||
```bash
|
||||
# Enable debug logging
|
||||
BASIC_MEMORY_LOG_LEVEL=DEBUG basic-memory sync
|
||||
|
||||
# View logs
|
||||
tail -f ~/.basic-memory/basic-memory.log
|
||||
|
||||
# Cloud/Docker mode (stdout logging with structured context)
|
||||
BASIC_MEMORY_CLOUD_MODE=true uvicorn basic_memory.api.app:app
|
||||
```
|
||||
|
||||
## Telemetry
|
||||
|
||||
Basic Memory collects anonymous usage statistics to help improve the software. This follows the [Homebrew model](https://docs.brew.sh/Analytics) - telemetry is on by default with easy opt-out.
|
||||
|
||||
**What we collect:**
|
||||
- App version, Python version, OS, architecture
|
||||
- Feature usage (which MCP tools and CLI commands are used)
|
||||
- Error types (sanitized - no file paths or personal data)
|
||||
|
||||
**What we NEVER collect:**
|
||||
- Note content, file names, or paths
|
||||
- Personal information
|
||||
- IP addresses
|
||||
|
||||
**Opting out:**
|
||||
```bash
|
||||
# Disable telemetry
|
||||
basic-memory telemetry disable
|
||||
|
||||
# Check status
|
||||
basic-memory telemetry status
|
||||
|
||||
# Re-enable
|
||||
basic-memory telemetry enable
|
||||
```
|
||||
|
||||
Or set the environment variable:
|
||||
```bash
|
||||
export BASIC_MEMORY_TELEMETRY_ENABLED=false
|
||||
```
|
||||
|
||||
For more details, see the [Telemetry documentation](https://basicmemory.com/telemetry).
|
||||
|
||||
## Development
|
||||
|
||||
### Running Tests
|
||||
|
||||
Basic Memory supports dual database backends (SQLite and Postgres). Tests are parametrized to run against both backends automatically.
|
||||
Basic Memory supports dual database backends (SQLite and Postgres). By default, tests run against SQLite. Set `BASIC_MEMORY_TEST_POSTGRES=1` to run against Postgres (uses testcontainers - Docker required).
|
||||
|
||||
**Quick Start:**
|
||||
```bash
|
||||
# Run SQLite tests (default, no Docker needed)
|
||||
# Run all tests against SQLite (default, fast)
|
||||
just test-sqlite
|
||||
|
||||
# Run Postgres tests (requires Docker)
|
||||
# Run all tests against Postgres (uses testcontainers)
|
||||
just test-postgres
|
||||
|
||||
# Run both SQLite and Postgres tests
|
||||
just test
|
||||
```
|
||||
|
||||
**Available Test Commands:**
|
||||
|
||||
- `just test-sqlite` - Run tests against SQLite only (fastest, no Docker needed)
|
||||
- `just test-postgres` - Run tests against Postgres only (requires Docker)
|
||||
- `just test` - Run all tests against both SQLite and Postgres
|
||||
- `just test-sqlite` - Run all tests against SQLite (fast, no Docker needed)
|
||||
- `just test-postgres` - Run all tests against Postgres (uses testcontainers)
|
||||
- `just test-unit-sqlite` - Run unit tests against SQLite
|
||||
- `just test-unit-postgres` - Run unit tests against Postgres
|
||||
- `just test-int-sqlite` - Run integration tests against SQLite
|
||||
- `just test-int-postgres` - Run integration tests against Postgres
|
||||
- `just test-windows` - Run Windows-specific tests (auto-skips on other platforms)
|
||||
- `just test-benchmark` - Run performance benchmark tests
|
||||
- `just test-all` - Run all tests including Windows, Postgres, and benchmarks
|
||||
|
||||
**Postgres Testing Requirements:**
|
||||
**Postgres Testing:**
|
||||
|
||||
To run Postgres tests, you need to start the test database:
|
||||
```bash
|
||||
docker-compose -f docker-compose-postgres.yml up -d
|
||||
```
|
||||
|
||||
Tests will connect to `localhost:5433/basic_memory_test`.
|
||||
Postgres tests use [testcontainers](https://testcontainers-python.readthedocs.io/) which automatically spins up a Postgres instance in Docker. No manual database setup required - just have Docker running.
|
||||
|
||||
**Test Markers:**
|
||||
|
||||
Tests use pytest markers for selective execution:
|
||||
- `postgres` - Tests that run against Postgres backend
|
||||
- `windows` - Windows-specific database optimizations
|
||||
- `benchmark` - Performance tests (excluded from default runs)
|
||||
|
||||
|
||||
@@ -7,44 +7,60 @@ install:
|
||||
@echo ""
|
||||
@echo "💡 Remember to activate the virtual environment by running: source .venv/bin/activate"
|
||||
|
||||
# Run all tests with unified coverage report
|
||||
test: test-unit test-int
|
||||
|
||||
# Run unit tests only (fast, no coverage)
|
||||
test-unit:
|
||||
uv run pytest -p pytest_mock -v --no-cov tests
|
||||
|
||||
# Run integration tests only (fast, no coverage)
|
||||
test-int:
|
||||
uv run pytest -p pytest_mock -v --no-cov test-int
|
||||
|
||||
# ==============================================================================
|
||||
# DATABASE BACKEND TESTING
|
||||
# ==============================================================================
|
||||
# Basic Memory supports dual database backends (SQLite and Postgres).
|
||||
# Tests are parametrized to run against both backends automatically.
|
||||
# By default, tests run against SQLite (fast, no dependencies).
|
||||
# Set BASIC_MEMORY_TEST_POSTGRES=1 to run against Postgres (uses testcontainers).
|
||||
#
|
||||
# Quick Start:
|
||||
# just test-sqlite # Run SQLite tests (default, no Docker needed)
|
||||
# just test-postgres # Run Postgres tests (requires Docker)
|
||||
# just test # Run all tests against SQLite (default)
|
||||
# just test-sqlite # Run all tests against SQLite
|
||||
# just test-postgres # Run all tests against Postgres (testcontainers)
|
||||
# just test-unit-sqlite # Run unit tests against SQLite
|
||||
# just test-unit-postgres # Run unit tests against Postgres
|
||||
# just test-int-sqlite # Run integration tests against SQLite
|
||||
# just test-int-postgres # Run integration tests against Postgres
|
||||
#
|
||||
# For Postgres tests, first start the database:
|
||||
# docker-compose -f docker-compose-postgres.yml up -d
|
||||
# CI runs both in parallel for faster feedback.
|
||||
# ==============================================================================
|
||||
|
||||
# Run tests against SQLite only (default backend, skip Postgres/Benchmark tests)
|
||||
# This is the fastest option and doesn't require any Docker setup.
|
||||
# Use this for local development and quick feedback.
|
||||
# Includes Windows-specific tests which will auto-skip on non-Windows platforms.
|
||||
test-sqlite:
|
||||
uv run pytest -p pytest_mock -v --no-cov -m "not postgres and not benchmark" tests test-int
|
||||
# Run all tests against SQLite and Postgres
|
||||
test: test-sqlite test-postgres
|
||||
|
||||
# Run tests against Postgres only (requires docker-compose-postgres.yml up)
|
||||
# First start Postgres: docker-compose -f docker-compose-postgres.yml up -d
|
||||
# Tests will connect to localhost:5433/basic_memory_test
|
||||
# To reset the database: just postgres-reset
|
||||
test-postgres:
|
||||
uv run pytest -p pytest_mock -v --no-cov -m "postgres and not benchmark" tests test-int
|
||||
# Run all tests against SQLite
|
||||
test-sqlite: test-unit-sqlite test-int-sqlite
|
||||
|
||||
# Run all tests against Postgres (uses testcontainers)
|
||||
test-postgres: test-unit-postgres test-int-postgres
|
||||
|
||||
# Run unit tests against SQLite
|
||||
test-unit-sqlite:
|
||||
BASIC_MEMORY_ENV=test uv run pytest -p pytest_mock -v --no-cov tests
|
||||
|
||||
# Run unit tests against Postgres
|
||||
test-unit-postgres:
|
||||
BASIC_MEMORY_ENV=test BASIC_MEMORY_TEST_POSTGRES=1 uv run pytest -p pytest_mock -v --no-cov tests
|
||||
|
||||
# Run integration tests against SQLite
|
||||
test-int-sqlite:
|
||||
uv run pytest -p pytest_mock -v --no-cov test-int
|
||||
|
||||
# Run integration tests against Postgres
|
||||
# Note: Uses timeout due to FastMCP Client + asyncpg cleanup hang (tests pass, process hangs on exit)
|
||||
# See: https://github.com/jlowin/fastmcp/issues/1311
|
||||
test-int-postgres:
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
# Use gtimeout (macOS/Homebrew) or timeout (Linux)
|
||||
TIMEOUT_CMD=$(command -v gtimeout || command -v timeout || echo "")
|
||||
if [[ -n "$TIMEOUT_CMD" ]]; then
|
||||
$TIMEOUT_CMD --signal=KILL 600 bash -c 'BASIC_MEMORY_TEST_POSTGRES=1 uv run pytest -p pytest_mock -v --no-cov test-int' || test $? -eq 137
|
||||
else
|
||||
echo "⚠️ No timeout command found, running without timeout..."
|
||||
BASIC_MEMORY_TEST_POSTGRES=1 uv run pytest -p pytest_mock -v --no-cov test-int
|
||||
fi
|
||||
|
||||
# Reset Postgres test database (drops and recreates schema)
|
||||
# Useful when Alembic migration state gets out of sync during development
|
||||
@@ -59,7 +75,7 @@ postgres-reset:
|
||||
postgres-migrate:
|
||||
@cd src/basic_memory/alembic && \
|
||||
BASIC_MEMORY_DATABASE_BACKEND=postgres \
|
||||
BASIC_MEMORY_DATABASE_URL=${POSTGRES_TEST_URL:-postgresql://basic_memory_user:dev_password@localhost:5433/basic_memory_test} \
|
||||
BASIC_MEMORY_DATABASE_URL=${POSTGRES_TEST_URL:-postgresql+asyncpg://basic_memory_user:dev_password@localhost:5433/basic_memory_test} \
|
||||
uv run alembic upgrade head
|
||||
@echo "✅ Migrations applied to Postgres test database"
|
||||
|
||||
|
||||
+10
-4
@@ -29,14 +29,19 @@ dependencies = [
|
||||
"alembic>=1.14.1",
|
||||
"pillow>=11.1.0",
|
||||
"pybars3>=0.9.7",
|
||||
"fastmcp>=2.10.2",
|
||||
"fastmcp==2.12.3", # Pinned - 2.14.x breaks MCP tools visibility (issue #463)
|
||||
"pyjwt>=2.10.1",
|
||||
"python-dotenv>=1.1.0",
|
||||
"pytest-aio>=1.9.0",
|
||||
"aiofiles>=24.1.0", # Async file I/O
|
||||
"logfire>=0.73.0", # Optional observability (disabled by default via config)
|
||||
"aiofiles>=24.1.0", # Optional observability (disabled by default via config)
|
||||
"asyncpg>=0.30.0",
|
||||
"nest-asyncio>=1.6.0", # For Alembic migrations with Postgres
|
||||
"pytest-asyncio>=1.2.0",
|
||||
"psycopg==3.3.1",
|
||||
"mdformat>=0.7.22",
|
||||
"mdformat-gfm>=0.3.7",
|
||||
"mdformat-frontmatter>=2.0.8",
|
||||
"openpanel>=0.0.1", # Anonymous usage telemetry (Homebrew-style opt-out)
|
||||
]
|
||||
|
||||
|
||||
@@ -81,7 +86,8 @@ dev = [
|
||||
"pytest-xdist>=3.0.0",
|
||||
"ruff>=0.1.6",
|
||||
"freezegun>=1.5.5",
|
||||
|
||||
"testcontainers[postgres]>=4.0.0",
|
||||
"psycopg>=3.2.0",
|
||||
]
|
||||
|
||||
[tool.hatch.version]
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""basic-memory - Local-first knowledge management combining Zettelkasten with knowledge graphs"""
|
||||
|
||||
# Package version - updated by release automation
|
||||
__version__ = "0.16.2"
|
||||
__version__ = "0.17.1"
|
||||
|
||||
# API version for FastAPI - independent of package version
|
||||
__api_version__ = "v0"
|
||||
|
||||
@@ -21,8 +21,12 @@ from alembic import context
|
||||
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
# set config.env to "test" for pytest to prevent logging to file in utils.setup_logging()
|
||||
os.environ["BASIC_MEMORY_ENV"] = "test"
|
||||
# Trigger: only set test env when actually running under pytest
|
||||
# Why: alembic/env.py is imported during normal operations (MCP server startup, migrations)
|
||||
# but we only want test behavior during actual test runs
|
||||
# Outcome: prevents is_test_env from returning True in production, enabling watch service
|
||||
if os.getenv("PYTEST_CURRENT_TEST") is not None:
|
||||
os.environ["BASIC_MEMORY_ENV"] = "test"
|
||||
|
||||
# Import after setting environment variable # noqa: E402
|
||||
from basic_memory.models import Base # noqa: E402
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Add cascade delete FK from search_index to entity
|
||||
|
||||
Revision ID: a2b3c4d5e6f7
|
||||
Revises: f8a9b2c3d4e5
|
||||
Create Date: 2025-12-02 07:00:00.000000
|
||||
|
||||
"""
|
||||
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "a2b3c4d5e6f7"
|
||||
down_revision: Union[str, None] = "f8a9b2c3d4e5"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add FK with CASCADE delete from search_index.entity_id to entity.id.
|
||||
|
||||
This migration is Postgres-only because:
|
||||
- SQLite uses FTS5 virtual tables which don't support foreign keys
|
||||
- The FK enables automatic cleanup of search_index entries when entities are deleted
|
||||
"""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
# First, clean up any orphaned search_index entries where entity no longer exists
|
||||
op.execute("""
|
||||
DELETE FROM search_index
|
||||
WHERE entity_id IS NOT NULL
|
||||
AND entity_id NOT IN (SELECT id FROM entity)
|
||||
""")
|
||||
|
||||
# Add FK with CASCADE - nullable FK allows search_index entries without entity_id
|
||||
op.create_foreign_key(
|
||||
"fk_search_index_entity_id",
|
||||
"search_index",
|
||||
"entity",
|
||||
["entity_id"],
|
||||
["id"],
|
||||
ondelete="CASCADE",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove the FK constraint."""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
op.drop_constraint("fk_search_index_entity_id", "search_index", type_="foreignkey")
|
||||
+239
@@ -0,0 +1,239 @@
|
||||
"""Add project_id to relation/observation and pg_trgm for fuzzy link resolution
|
||||
|
||||
Revision ID: f8a9b2c3d4e5
|
||||
Revises: 314f1ea54dc4
|
||||
Create Date: 2025-12-01 12:00:00.000000
|
||||
|
||||
"""
|
||||
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
|
||||
def column_exists(connection, table: str, column: str) -> bool:
|
||||
"""Check if a column exists in a table (idempotent migration support)."""
|
||||
if connection.dialect.name == "postgresql":
|
||||
result = connection.execute(
|
||||
text(
|
||||
"SELECT 1 FROM information_schema.columns "
|
||||
"WHERE table_name = :table AND column_name = :column"
|
||||
),
|
||||
{"table": table, "column": column},
|
||||
)
|
||||
return result.fetchone() is not None
|
||||
else:
|
||||
# SQLite
|
||||
result = connection.execute(text(f"PRAGMA table_info({table})"))
|
||||
columns = [row[1] for row in result]
|
||||
return column in columns
|
||||
|
||||
|
||||
def index_exists(connection, index_name: str) -> bool:
|
||||
"""Check if an index exists (idempotent migration support)."""
|
||||
if connection.dialect.name == "postgresql":
|
||||
result = connection.execute(
|
||||
text("SELECT 1 FROM pg_indexes WHERE indexname = :index_name"),
|
||||
{"index_name": index_name},
|
||||
)
|
||||
return result.fetchone() is not None
|
||||
else:
|
||||
# SQLite
|
||||
result = connection.execute(
|
||||
text("SELECT 1 FROM sqlite_master WHERE type='index' AND name = :index_name"),
|
||||
{"index_name": index_name},
|
||||
)
|
||||
return result.fetchone() is not None
|
||||
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "f8a9b2c3d4e5"
|
||||
down_revision: Union[str, None] = "314f1ea54dc4"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add project_id to relation and observation tables, plus pg_trgm indexes.
|
||||
|
||||
This migration:
|
||||
1. Adds project_id column to relation and observation tables (denormalization)
|
||||
2. Backfills project_id from the associated entity
|
||||
3. Enables pg_trgm extension for trigram-based fuzzy matching (Postgres only)
|
||||
4. Creates GIN indexes on entity title and permalink for fast similarity searches
|
||||
5. Creates partial index on unresolved relations for efficient bulk resolution
|
||||
"""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Add project_id to relation table
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
# Step 1: Add project_id column as nullable first (idempotent)
|
||||
if not column_exists(connection, "relation", "project_id"):
|
||||
op.add_column("relation", sa.Column("project_id", sa.Integer(), nullable=True))
|
||||
|
||||
# Step 2: Backfill project_id from entity.project_id via from_id
|
||||
if dialect == "postgresql":
|
||||
op.execute("""
|
||||
UPDATE relation
|
||||
SET project_id = entity.project_id
|
||||
FROM entity
|
||||
WHERE relation.from_id = entity.id
|
||||
""")
|
||||
else:
|
||||
# SQLite syntax
|
||||
op.execute("""
|
||||
UPDATE relation
|
||||
SET project_id = (
|
||||
SELECT entity.project_id
|
||||
FROM entity
|
||||
WHERE entity.id = relation.from_id
|
||||
)
|
||||
""")
|
||||
|
||||
# Step 3: Make project_id NOT NULL and add foreign key
|
||||
if dialect == "postgresql":
|
||||
op.alter_column("relation", "project_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_relation_project_id",
|
||||
"relation",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
else:
|
||||
# SQLite requires batch operations for ALTER COLUMN
|
||||
with op.batch_alter_table("relation") as batch_op:
|
||||
batch_op.alter_column("project_id", nullable=False)
|
||||
batch_op.create_foreign_key(
|
||||
"fk_relation_project_id",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
|
||||
# Step 4: Create index on relation.project_id (idempotent)
|
||||
if not index_exists(connection, "ix_relation_project_id"):
|
||||
op.create_index("ix_relation_project_id", "relation", ["project_id"])
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Add project_id to observation table
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
# Step 1: Add project_id column as nullable first (idempotent)
|
||||
if not column_exists(connection, "observation", "project_id"):
|
||||
op.add_column("observation", sa.Column("project_id", sa.Integer(), nullable=True))
|
||||
|
||||
# Step 2: Backfill project_id from entity.project_id via entity_id
|
||||
if dialect == "postgresql":
|
||||
op.execute("""
|
||||
UPDATE observation
|
||||
SET project_id = entity.project_id
|
||||
FROM entity
|
||||
WHERE observation.entity_id = entity.id
|
||||
""")
|
||||
else:
|
||||
# SQLite syntax
|
||||
op.execute("""
|
||||
UPDATE observation
|
||||
SET project_id = (
|
||||
SELECT entity.project_id
|
||||
FROM entity
|
||||
WHERE entity.id = observation.entity_id
|
||||
)
|
||||
""")
|
||||
|
||||
# Step 3: Make project_id NOT NULL and add foreign key
|
||||
if dialect == "postgresql":
|
||||
op.alter_column("observation", "project_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_observation_project_id",
|
||||
"observation",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
else:
|
||||
# SQLite requires batch operations for ALTER COLUMN
|
||||
with op.batch_alter_table("observation") as batch_op:
|
||||
batch_op.alter_column("project_id", nullable=False)
|
||||
batch_op.create_foreign_key(
|
||||
"fk_observation_project_id",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
|
||||
# Step 4: Create index on observation.project_id (idempotent)
|
||||
if not index_exists(connection, "ix_observation_project_id"):
|
||||
op.create_index("ix_observation_project_id", "observation", ["project_id"])
|
||||
|
||||
# Postgres-specific: pg_trgm and GIN indexes
|
||||
if dialect == "postgresql":
|
||||
# Enable pg_trgm extension for fuzzy string matching
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS pg_trgm")
|
||||
|
||||
# Create trigram indexes on entity table for fuzzy matching
|
||||
# GIN indexes with gin_trgm_ops support similarity searches
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_entity_title_trgm
|
||||
ON entity USING gin (title gin_trgm_ops)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_entity_permalink_trgm
|
||||
ON entity USING gin (permalink gin_trgm_ops)
|
||||
""")
|
||||
|
||||
# Create partial index on unresolved relations for efficient bulk resolution
|
||||
# This makes "WHERE to_id IS NULL AND project_id = X" queries very fast
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_relation_unresolved
|
||||
ON relation (project_id, to_name)
|
||||
WHERE to_id IS NULL
|
||||
""")
|
||||
|
||||
# Create index on relation.to_name for join performance in bulk resolution
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_relation_to_name
|
||||
ON relation (to_name)
|
||||
""")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove project_id from relation/observation and pg_trgm indexes."""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
# Drop Postgres-specific indexes
|
||||
op.execute("DROP INDEX IF EXISTS idx_relation_to_name")
|
||||
op.execute("DROP INDEX IF EXISTS idx_relation_unresolved")
|
||||
op.execute("DROP INDEX IF EXISTS idx_entity_permalink_trgm")
|
||||
op.execute("DROP INDEX IF EXISTS idx_entity_title_trgm")
|
||||
# Note: We don't drop the pg_trgm extension as other code may depend on it
|
||||
|
||||
# Drop project_id from observation
|
||||
op.drop_index("ix_observation_project_id", table_name="observation")
|
||||
op.drop_constraint("fk_observation_project_id", "observation", type_="foreignkey")
|
||||
op.drop_column("observation", "project_id")
|
||||
|
||||
# Drop project_id from relation
|
||||
op.drop_index("ix_relation_project_id", table_name="relation")
|
||||
op.drop_constraint("fk_relation_project_id", "relation", type_="foreignkey")
|
||||
op.drop_column("relation", "project_id")
|
||||
else:
|
||||
# SQLite requires batch operations
|
||||
op.drop_index("ix_observation_project_id", table_name="observation")
|
||||
with op.batch_alter_table("observation") as batch_op:
|
||||
batch_op.drop_constraint("fk_observation_project_id", type_="foreignkey")
|
||||
batch_op.drop_column("project_id")
|
||||
|
||||
op.drop_index("ix_relation_project_id", table_name="relation")
|
||||
with op.batch_alter_table("relation") as batch_op:
|
||||
batch_op.drop_constraint("fk_relation_project_id", type_="foreignkey")
|
||||
batch_op.drop_column("project_id")
|
||||
@@ -30,7 +30,7 @@ from basic_memory.api.v2.routers import (
|
||||
prompt_router as v2_prompt,
|
||||
importer_router as v2_importer,
|
||||
)
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.config import ConfigManager, init_api_logging
|
||||
from basic_memory.services.initialization import initialize_file_sync, initialize_app
|
||||
|
||||
|
||||
@@ -38,6 +38,9 @@ from basic_memory.services.initialization import initialize_file_sync, initializ
|
||||
async def lifespan(app: FastAPI): # pragma: no cover
|
||||
"""Lifecycle manager for the FastAPI app. Not called in stdio mcp mode"""
|
||||
|
||||
# Initialize logging for API (stdout in cloud mode, file otherwise)
|
||||
init_api_logging()
|
||||
|
||||
app_config = ConfigManager().config
|
||||
logger.info("Starting Basic Memory API")
|
||||
|
||||
@@ -50,12 +53,21 @@ async def lifespan(app: FastAPI): # pragma: no cover
|
||||
app.state.session_maker = session_maker
|
||||
logger.info("Database connections cached in app state")
|
||||
|
||||
logger.info(f"Sync changes enabled: {app_config.sync_changes}")
|
||||
if app_config.sync_changes:
|
||||
# Start file sync if enabled
|
||||
if app_config.sync_changes and not app_config.is_test_env:
|
||||
logger.info(f"Sync changes enabled: {app_config.sync_changes}")
|
||||
|
||||
# start file sync task in background
|
||||
app.state.sync_task = asyncio.create_task(initialize_file_sync(app_config))
|
||||
async def _file_sync_runner() -> None:
|
||||
await initialize_file_sync(app_config)
|
||||
|
||||
app.state.sync_task = asyncio.create_task(_file_sync_runner())
|
||||
else:
|
||||
logger.info("Sync changes disabled. Skipping file sync service.")
|
||||
if app_config.is_test_env:
|
||||
logger.info("Test environment detected. Skipping file sync service.")
|
||||
else:
|
||||
logger.info("Sync changes disabled. Skipping file sync service.")
|
||||
app.state.sync_task = None
|
||||
|
||||
# proceed with startup
|
||||
yield
|
||||
@@ -64,6 +76,10 @@ async def lifespan(app: FastAPI): # pragma: no cover
|
||||
if app.state.sync_task:
|
||||
logger.info("Stopping sync...")
|
||||
app.state.sync_task.cancel() # pyright: ignore
|
||||
try:
|
||||
await app.state.sync_task
|
||||
except asyncio.CancelledError:
|
||||
logger.info("Sync task cancelled successfully")
|
||||
|
||||
await db.shutdown_db()
|
||||
|
||||
@@ -100,8 +116,6 @@ app.include_router(v2_project, prefix="/v2")
|
||||
app.include_router(project.project_resource_router)
|
||||
app.include_router(management.router)
|
||||
|
||||
# Auth routes are handled by FastMCP automatically when auth is enabled
|
||||
|
||||
|
||||
@app.exception_handler(Exception)
|
||||
async def exception_handler(request, exc): # pragma: no cover
|
||||
|
||||
@@ -274,7 +274,7 @@ async def add_project(
|
||||
raise HTTPException(status_code=500, detail="Failed to retrieve newly created project")
|
||||
|
||||
return ProjectStatusResponse( # pyright: ignore [reportCallIssue]
|
||||
message=f"Project '{project_data.name}' added successfully",
|
||||
message=f"Project '{new_project.name}' added successfully",
|
||||
status="success",
|
||||
default=project_data.set_default,
|
||||
new_project=ProjectItem(
|
||||
@@ -329,7 +329,7 @@ async def remove_project(
|
||||
await project_service.remove_project(name, delete_notes=delete_notes)
|
||||
|
||||
return ProjectStatusResponse(
|
||||
message=f"Project '{name}' removed successfully",
|
||||
message=f"Project '{old_project.name}' removed successfully",
|
||||
status="success",
|
||||
default=False,
|
||||
old_project=ProjectItem(
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
from typing import Annotated, Union
|
||||
|
||||
from fastapi import APIRouter, HTTPException, BackgroundTasks, Body
|
||||
from fastapi import APIRouter, HTTPException, BackgroundTasks, Body, Response
|
||||
from fastapi.responses import FileResponse, JSONResponse
|
||||
from loguru import logger
|
||||
|
||||
@@ -25,6 +25,17 @@ from datetime import datetime
|
||||
router = APIRouter(prefix="/resource", tags=["resources"])
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: EntityModel) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
def get_entity_ids(item: SearchIndexRow) -> set[int]:
|
||||
match item.type:
|
||||
case SearchItemType.ENTITY:
|
||||
@@ -39,7 +50,7 @@ def get_entity_ids(item: SearchIndexRow) -> set[int]:
|
||||
raise ValueError(f"Unexpected type: {item.type}")
|
||||
|
||||
|
||||
@router.get("/{identifier:path}")
|
||||
@router.get("/{identifier:path}", response_model=None)
|
||||
async def get_resource_content(
|
||||
config: ProjectConfigDep,
|
||||
link_resolver: LinkResolverDep,
|
||||
@@ -50,7 +61,7 @@ async def get_resource_content(
|
||||
identifier: str,
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
) -> FileResponse:
|
||||
) -> Union[Response, FileResponse]:
|
||||
"""Get resource content by identifier: name or permalink."""
|
||||
logger.debug(f"Getting content for: {identifier}")
|
||||
|
||||
@@ -81,13 +92,16 @@ async def get_resource_content(
|
||||
# return single response
|
||||
if len(results) == 1:
|
||||
entity = results[0]
|
||||
file_path = Path(f"{config.home}/{entity.file_path}")
|
||||
if not file_path.exists():
|
||||
# Check file exists via file_service (for cloud compatibility)
|
||||
if not await file_service.exists(entity.file_path):
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"File not found: {file_path}",
|
||||
detail=f"File not found: {entity.file_path}",
|
||||
)
|
||||
return FileResponse(path=file_path)
|
||||
# Read content via file_service as bytes (works with both local and S3)
|
||||
content = await file_service.read_file_bytes(entity.file_path)
|
||||
content_type = file_service.content_type(entity.file_path)
|
||||
return Response(content=content, media_type=content_type)
|
||||
|
||||
# for multiple files, initialize a temporary file for writing the results
|
||||
with tempfile.NamedTemporaryFile(delete=False, mode="w", suffix=".md") as tmp_file:
|
||||
@@ -97,7 +111,7 @@ async def get_resource_content(
|
||||
# Read content for each entity
|
||||
content = await file_service.read_entity_content(result)
|
||||
memory_url = normalize_memory_url(result.permalink)
|
||||
modified_date = result.updated_at.isoformat()
|
||||
modified_date = _mtime_to_datetime(result).isoformat()
|
||||
checksum = result.checksum[:8] if result.checksum else ""
|
||||
|
||||
# Prepare the delimited content
|
||||
@@ -171,21 +185,17 @@ async def write_resource(
|
||||
else:
|
||||
content_str = str(content)
|
||||
|
||||
# Get full file path
|
||||
full_path = Path(f"{config.home}/{file_path}")
|
||||
|
||||
# Ensure parent directory exists
|
||||
full_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Write content to file
|
||||
checksum = await file_service.write_file(full_path, content_str)
|
||||
# Cloud compatibility: do not assume a local filesystem path structure.
|
||||
# Delegate directory creation + writes to the configured FileService (local or S3).
|
||||
await file_service.ensure_directory(Path(file_path).parent)
|
||||
checksum = await file_service.write_file(file_path, content_str)
|
||||
|
||||
# Get file info
|
||||
file_stats = file_service.file_stats(full_path)
|
||||
file_metadata = await file_service.get_file_metadata(file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(file_path).name
|
||||
content_type = file_service.content_type(full_path)
|
||||
content_type = file_service.content_type(file_path)
|
||||
|
||||
entity_type = "canvas" if file_path.endswith(".canvas") else "file"
|
||||
|
||||
@@ -202,7 +212,7 @@ async def write_resource(
|
||||
"content_type": content_type,
|
||||
"file_path": file_path,
|
||||
"checksum": checksum,
|
||||
"updated_at": datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
"updated_at": file_metadata.modified_at,
|
||||
},
|
||||
)
|
||||
status_code = 200
|
||||
@@ -214,8 +224,8 @@ async def write_resource(
|
||||
content_type=content_type,
|
||||
file_path=file_path,
|
||||
checksum=checksum,
|
||||
created_at=datetime.fromtimestamp(file_stats.st_ctime).astimezone(),
|
||||
updated_at=datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
created_at=file_metadata.created_at,
|
||||
updated_at=file_metadata.modified_at,
|
||||
)
|
||||
entity = await entity_repository.add(entity)
|
||||
status_code = 201
|
||||
@@ -229,9 +239,9 @@ async def write_resource(
|
||||
content={
|
||||
"file_path": file_path,
|
||||
"checksum": checksum,
|
||||
"size": file_stats.st_size,
|
||||
"created_at": file_stats.st_ctime,
|
||||
"modified_at": file_stats.st_mtime,
|
||||
"size": file_metadata.size,
|
||||
"created_at": file_metadata.created_at.timestamp(),
|
||||
"modified_at": file_metadata.modified_at.timestamp(),
|
||||
},
|
||||
)
|
||||
except Exception as e: # pragma: no cover
|
||||
|
||||
@@ -24,8 +24,26 @@ async def to_graph_context(
|
||||
page: Optional[int] = None,
|
||||
page_size: Optional[int] = None,
|
||||
):
|
||||
# First pass: collect all entity IDs needed for relations
|
||||
entity_ids_needed: set[int] = set()
|
||||
for context_item in context_result.results:
|
||||
for item in (
|
||||
[context_item.primary_result] + context_item.observations + context_item.related_results
|
||||
):
|
||||
if item.type == SearchItemType.RELATION:
|
||||
if item.from_id: # pyright: ignore
|
||||
entity_ids_needed.add(item.from_id) # pyright: ignore
|
||||
if item.to_id:
|
||||
entity_ids_needed.add(item.to_id)
|
||||
|
||||
# Batch fetch all entities at once
|
||||
entity_lookup: dict[int, str] = {}
|
||||
if entity_ids_needed:
|
||||
entities = await entity_repository.find_by_ids(list(entity_ids_needed))
|
||||
entity_lookup = {e.id: e.title for e in entities}
|
||||
|
||||
# Helper function to convert items to summaries
|
||||
async def to_summary(item: SearchIndexRow | ContextResultRow):
|
||||
def to_summary(item: SearchIndexRow | ContextResultRow):
|
||||
match item.type:
|
||||
case SearchItemType.ENTITY:
|
||||
return EntitySummary(
|
||||
@@ -48,8 +66,8 @@ async def to_graph_context(
|
||||
created_at=item.created_at,
|
||||
)
|
||||
case SearchItemType.RELATION:
|
||||
from_entity = await entity_repository.find_by_id(item.from_id) # pyright: ignore
|
||||
to_entity = await entity_repository.find_by_id(item.to_id) if item.to_id else None
|
||||
from_title = entity_lookup.get(item.from_id) if item.from_id else None # pyright: ignore
|
||||
to_title = entity_lookup.get(item.to_id) if item.to_id else None
|
||||
return RelationSummary(
|
||||
relation_id=item.id,
|
||||
entity_id=item.entity_id, # pyright: ignore
|
||||
@@ -57,9 +75,9 @@ async def to_graph_context(
|
||||
file_path=item.file_path,
|
||||
permalink=item.permalink, # pyright: ignore
|
||||
relation_type=item.relation_type, # pyright: ignore
|
||||
from_entity=from_entity.title if from_entity else None,
|
||||
from_entity=from_title,
|
||||
from_entity_id=item.from_id, # pyright: ignore
|
||||
to_entity=to_entity.title if to_entity else None,
|
||||
to_entity=to_title,
|
||||
to_entity_id=item.to_id,
|
||||
created_at=item.created_at,
|
||||
)
|
||||
@@ -70,23 +88,19 @@ async def to_graph_context(
|
||||
hierarchical_results = []
|
||||
for context_item in context_result.results:
|
||||
# Process primary result
|
||||
primary_result = await to_summary(context_item.primary_result)
|
||||
primary_result = to_summary(context_item.primary_result)
|
||||
|
||||
# Process observations
|
||||
observations = []
|
||||
for obs in context_item.observations:
|
||||
observations.append(await to_summary(obs))
|
||||
# Process observations (always ObservationSummary, validated by context_service)
|
||||
observations = [to_summary(obs) for obs in context_item.observations]
|
||||
|
||||
# Process related results
|
||||
related = []
|
||||
for rel in context_item.related_results:
|
||||
related.append(await to_summary(rel))
|
||||
related = [to_summary(rel) for rel in context_item.related_results]
|
||||
|
||||
# Add to hierarchical results
|
||||
hierarchical_results.append(
|
||||
ContextResult(
|
||||
primary_result=primary_result,
|
||||
observations=observations,
|
||||
observations=observations, # pyright: ignore[reportArgumentType]
|
||||
related_results=related,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -96,9 +96,7 @@ async def resolve_identifier(
|
||||
# Try to resolve the identifier
|
||||
entity = await link_resolver.resolve_link(data.identifier)
|
||||
if not entity:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Could not resolve identifier: '{data.identifier}'"
|
||||
)
|
||||
raise HTTPException(status_code=404, detail=f"Entity not found: '{data.identifier}'")
|
||||
|
||||
# Determine resolution method
|
||||
resolution_method = "search" # default
|
||||
|
||||
@@ -25,11 +25,89 @@ from basic_memory.schemas.project_info import (
|
||||
ProjectItem,
|
||||
ProjectStatusResponse,
|
||||
)
|
||||
from basic_memory.utils import normalize_project_path
|
||||
from basic_memory.schemas.v2 import ProjectResolveRequest, ProjectResolveResponse
|
||||
from basic_memory.utils import normalize_project_path, generate_permalink
|
||||
|
||||
router = APIRouter(prefix="/projects", tags=["project_management-v2"])
|
||||
|
||||
|
||||
@router.post("/resolve", response_model=ProjectResolveResponse)
|
||||
async def resolve_project_identifier(
|
||||
data: ProjectResolveRequest,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
) -> ProjectResolveResponse:
|
||||
"""Resolve a project identifier (name or permalink) to a project ID.
|
||||
|
||||
This endpoint provides efficient lookup of projects by name without
|
||||
needing to fetch the entire project list. Supports case-insensitive
|
||||
matching on both name and permalink.
|
||||
|
||||
Args:
|
||||
data: Request containing the identifier to resolve
|
||||
|
||||
Returns:
|
||||
Project information including the numeric ID
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if project not found
|
||||
|
||||
Example:
|
||||
POST /v2/projects/resolve
|
||||
{"identifier": "my-project"}
|
||||
|
||||
Returns:
|
||||
{
|
||||
"project_id": 1,
|
||||
"name": "my-project",
|
||||
"permalink": "my-project",
|
||||
"path": "/path/to/project",
|
||||
"is_active": true,
|
||||
"is_default": false,
|
||||
"resolution_method": "name"
|
||||
}
|
||||
"""
|
||||
logger.info(f"API v2 request: resolve_project_identifier for '{data.identifier}'")
|
||||
|
||||
# Generate permalink for comparison
|
||||
identifier_permalink = generate_permalink(data.identifier)
|
||||
|
||||
# Try to find project by ID first (if identifier is numeric)
|
||||
resolution_method = "name"
|
||||
project = None
|
||||
|
||||
if data.identifier.isdigit():
|
||||
project_id = int(data.identifier)
|
||||
project = await project_repository.get_by_id(project_id)
|
||||
if project:
|
||||
resolution_method = "id"
|
||||
|
||||
# If not found by ID, try by permalink first (exact match)
|
||||
if not project:
|
||||
project = await project_repository.get_by_permalink(identifier_permalink)
|
||||
if project:
|
||||
resolution_method = "permalink"
|
||||
|
||||
# If not found by permalink, try case-insensitive name search
|
||||
# Uses efficient database query instead of fetching all projects
|
||||
if not project:
|
||||
project = await project_repository.get_by_name_case_insensitive(data.identifier)
|
||||
if project:
|
||||
resolution_method = "name"
|
||||
|
||||
if not project:
|
||||
raise HTTPException(status_code=404, detail=f"Project not found: '{data.identifier}'")
|
||||
|
||||
return ProjectResolveResponse(
|
||||
project_id=project.id,
|
||||
name=project.name,
|
||||
permalink=generate_permalink(project.name),
|
||||
path=normalize_project_path(project.path),
|
||||
is_active=project.is_active if hasattr(project, "is_active") else True,
|
||||
is_default=project.is_default or False,
|
||||
resolution_method=resolution_method,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/{project_id}", response_model=ProjectItem)
|
||||
async def get_project_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
|
||||
@@ -11,8 +11,7 @@ Key differences from v1:
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from fastapi.responses import FileResponse
|
||||
from fastapi import APIRouter, HTTPException, Response
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.deps import (
|
||||
@@ -30,7 +29,6 @@ from basic_memory.schemas.v2.resource import (
|
||||
ResourceResponse,
|
||||
)
|
||||
from basic_memory.utils import validate_project_path
|
||||
from datetime import datetime
|
||||
|
||||
router = APIRouter(prefix="/resource", tags=["resources-v2"])
|
||||
|
||||
@@ -42,7 +40,7 @@ async def get_resource_content(
|
||||
config: ProjectConfigV2Dep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
) -> FileResponse:
|
||||
) -> Response:
|
||||
"""Get raw resource content by entity ID.
|
||||
|
||||
Args:
|
||||
@@ -53,7 +51,7 @@ async def get_resource_content(
|
||||
file_service: File service for reading file content
|
||||
|
||||
Returns:
|
||||
FileResponse with entity content
|
||||
Response with entity content
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if entity or file not found
|
||||
@@ -76,14 +74,18 @@ async def get_resource_content(
|
||||
detail="Entity contains invalid file path",
|
||||
)
|
||||
|
||||
file_path = Path(f"{config.home}/{entity.file_path}")
|
||||
if not file_path.exists():
|
||||
# Check file exists via file_service (for cloud compatibility)
|
||||
if not await file_service.exists(entity.file_path):
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"File not found: {file_path}",
|
||||
detail=f"File not found: {entity.file_path}",
|
||||
)
|
||||
|
||||
return FileResponse(path=file_path)
|
||||
# Read content via file_service as bytes (works with both local and S3)
|
||||
content = await file_service.read_file_bytes(entity.file_path)
|
||||
content_type = file_service.content_type(entity.file_path)
|
||||
|
||||
return Response(content=content, media_type=content_type)
|
||||
|
||||
|
||||
@router.post("", response_model=ResourceResponse)
|
||||
@@ -133,21 +135,17 @@ async def create_resource(
|
||||
f"Use PUT /resource/{existing_entity.id} to update it.",
|
||||
)
|
||||
|
||||
# Get full file path
|
||||
full_path = Path(f"{config.home}/{data.file_path}")
|
||||
|
||||
# Ensure parent directory exists
|
||||
full_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Write content to file
|
||||
checksum = await file_service.write_file(full_path, data.content)
|
||||
# Cloud compatibility: avoid assuming a local filesystem path.
|
||||
# Delegate directory creation + writes to FileService (local or S3).
|
||||
await file_service.ensure_directory(Path(data.file_path).parent)
|
||||
checksum = await file_service.write_file(data.file_path, data.content)
|
||||
|
||||
# Get file info
|
||||
file_stats = file_service.file_stats(full_path)
|
||||
file_metadata = await file_service.get_file_metadata(data.file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(data.file_path).name
|
||||
content_type = file_service.content_type(full_path)
|
||||
content_type = file_service.content_type(data.file_path)
|
||||
entity_type = "canvas" if data.file_path.endswith(".canvas") else "file"
|
||||
|
||||
# Create a new entity model
|
||||
@@ -157,8 +155,8 @@ async def create_resource(
|
||||
content_type=content_type,
|
||||
file_path=data.file_path,
|
||||
checksum=checksum,
|
||||
created_at=datetime.fromtimestamp(file_stats.st_ctime).astimezone(),
|
||||
updated_at=datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
created_at=file_metadata.created_at,
|
||||
updated_at=file_metadata.modified_at,
|
||||
)
|
||||
entity = await entity_repository.add(entity)
|
||||
|
||||
@@ -170,9 +168,9 @@ async def create_resource(
|
||||
entity_id=entity.id,
|
||||
file_path=data.file_path,
|
||||
checksum=checksum,
|
||||
size=file_stats.st_size,
|
||||
created_at=file_stats.st_ctime,
|
||||
modified_at=file_stats.st_mtime,
|
||||
size=file_metadata.size,
|
||||
created_at=file_metadata.created_at.timestamp(),
|
||||
modified_at=file_metadata.modified_at.timestamp(),
|
||||
)
|
||||
except HTTPException:
|
||||
# Re-raise HTTP exceptions without wrapping
|
||||
@@ -232,31 +230,27 @@ async def update_resource(
|
||||
"Path must be relative and stay within project boundaries.",
|
||||
)
|
||||
|
||||
# Get full paths
|
||||
old_full_path = Path(f"{config.home}/{entity.file_path}")
|
||||
new_full_path = Path(f"{config.home}/{target_file_path}")
|
||||
|
||||
# If moving file, handle the move
|
||||
if data.file_path and data.file_path != entity.file_path:
|
||||
# Ensure new parent directory exists
|
||||
new_full_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
# Ensure new parent directory exists (no-op for S3)
|
||||
await file_service.ensure_directory(Path(target_file_path).parent)
|
||||
|
||||
# If old file exists, remove it
|
||||
if old_full_path.exists():
|
||||
old_full_path.unlink()
|
||||
# If old file exists, remove it via file_service (for cloud compatibility)
|
||||
if await file_service.exists(entity.file_path):
|
||||
await file_service.delete_file(entity.file_path)
|
||||
else:
|
||||
# Ensure directory exists for in-place update
|
||||
new_full_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
await file_service.ensure_directory(Path(target_file_path).parent)
|
||||
|
||||
# Write content to target file
|
||||
checksum = await file_service.write_file(new_full_path, data.content)
|
||||
checksum = await file_service.write_file(target_file_path, data.content)
|
||||
|
||||
# Get file info
|
||||
file_stats = file_service.file_stats(new_full_path)
|
||||
file_metadata = await file_service.get_file_metadata(target_file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(target_file_path).name
|
||||
content_type = file_service.content_type(new_full_path)
|
||||
content_type = file_service.content_type(target_file_path)
|
||||
entity_type = "canvas" if target_file_path.endswith(".canvas") else "file"
|
||||
|
||||
# Update entity
|
||||
@@ -268,7 +262,7 @@ async def update_resource(
|
||||
"content_type": content_type,
|
||||
"file_path": target_file_path,
|
||||
"checksum": checksum,
|
||||
"updated_at": datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
"updated_at": file_metadata.modified_at,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -280,9 +274,9 @@ async def update_resource(
|
||||
entity_id=entity_id,
|
||||
file_path=target_file_path,
|
||||
checksum=checksum,
|
||||
size=file_stats.st_size,
|
||||
created_at=file_stats.st_ctime,
|
||||
modified_at=file_stats.st_mtime,
|
||||
size=file_metadata.size,
|
||||
created_at=file_metadata.created_at.timestamp(),
|
||||
modified_at=file_metadata.modified_at.timestamp(),
|
||||
)
|
||||
except HTTPException:
|
||||
# Re-raise HTTP exceptions without wrapping
|
||||
|
||||
@@ -1,8 +1,21 @@
|
||||
from typing import Optional
|
||||
# Suppress Logfire "not configured" warning - we only use Logfire in cloud/server contexts
|
||||
import os
|
||||
|
||||
import typer
|
||||
os.environ.setdefault("LOGFIRE_IGNORE_NO_CONFIG", "1")
|
||||
|
||||
from basic_memory.config import ConfigManager
|
||||
# Remove loguru's default handler IMMEDIATELY, before any other imports.
|
||||
# This prevents DEBUG logs from appearing on stdout during module-level
|
||||
# initialization (e.g., template_loader.TemplateLoader() logs at DEBUG level).
|
||||
from loguru import logger
|
||||
|
||||
logger.remove()
|
||||
|
||||
from typing import Optional # noqa: E402
|
||||
|
||||
import typer # noqa: E402
|
||||
|
||||
from basic_memory.config import ConfigManager, init_cli_logging # noqa: E402
|
||||
from basic_memory.telemetry import show_notice_if_needed, track_app_started # noqa: E402
|
||||
|
||||
|
||||
def version_callback(value: bool) -> None:
|
||||
@@ -31,8 +44,25 @@ def app_callback(
|
||||
) -> None:
|
||||
"""Basic Memory - Local-first personal knowledge management."""
|
||||
|
||||
# Run initialization for every command unless --version was specified
|
||||
if not version and ctx.invoked_subcommand is not None:
|
||||
# Initialize logging for CLI (file only, no stdout)
|
||||
init_cli_logging()
|
||||
|
||||
# Show telemetry notice and track CLI startup
|
||||
# Skip for 'mcp' command - it handles its own telemetry in lifespan
|
||||
# Skip for 'telemetry' command - avoid issues when user is managing telemetry
|
||||
if ctx.invoked_subcommand not in {"mcp", "telemetry"}:
|
||||
show_notice_if_needed()
|
||||
track_app_started("cli")
|
||||
|
||||
# Run initialization for commands that don't use the API
|
||||
# Skip for 'mcp' command - it has its own lifespan that handles initialization
|
||||
# Skip for API-using commands (status, sync, etc.) - they handle initialization via deps.py
|
||||
api_commands = {"mcp", "status", "sync", "project", "tool"}
|
||||
if (
|
||||
not version
|
||||
and ctx.invoked_subcommand is not None
|
||||
and ctx.invoked_subcommand not in api_commands
|
||||
):
|
||||
from basic_memory.services.initialization import ensure_initialization
|
||||
|
||||
app_config = ConfigManager().config
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""CLI commands for basic-memory."""
|
||||
|
||||
from . import status, db, import_memory_json, mcp, import_claude_conversations
|
||||
from . import import_claude_projects, import_chatgpt, tool, project
|
||||
from . import import_claude_projects, import_chatgpt, tool, project, format, telemetry
|
||||
|
||||
__all__ = [
|
||||
"status",
|
||||
@@ -13,4 +13,6 @@ __all__ = [
|
||||
"import_chatgpt",
|
||||
"tool",
|
||||
"project",
|
||||
"format",
|
||||
"telemetry",
|
||||
]
|
||||
|
||||
@@ -9,11 +9,14 @@ This module provides simplified, project-scoped rclone operations:
|
||||
Replaces tenant-wide sync with project-scoped workflows.
|
||||
"""
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from loguru import logger
|
||||
from rich.console import Console
|
||||
|
||||
from basic_memory.cli.commands.cloud.rclone_installer import is_rclone_installed
|
||||
@@ -21,6 +24,9 @@ from basic_memory.utils import normalize_project_path
|
||||
|
||||
console = Console()
|
||||
|
||||
# Minimum rclone version for --create-empty-src-dirs support
|
||||
MIN_RCLONE_VERSION_EMPTY_DIRS = (1, 64, 0)
|
||||
|
||||
|
||||
class RcloneError(Exception):
|
||||
"""Exception raised for rclone command errors."""
|
||||
@@ -43,6 +49,42 @@ def check_rclone_installed() -> None:
|
||||
)
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_rclone_version() -> tuple[int, int, int] | None:
|
||||
"""Get rclone version as (major, minor, patch) tuple.
|
||||
|
||||
Returns:
|
||||
Version tuple like (1, 64, 2), or None if version cannot be determined.
|
||||
|
||||
Note:
|
||||
Result is cached since rclone version won't change during runtime.
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(["rclone", "version"], capture_output=True, text=True, timeout=10)
|
||||
# Parse "rclone v1.64.2" or "rclone v1.60.1-DEV"
|
||||
match = re.search(r"v(\d+)\.(\d+)\.(\d+)", result.stdout)
|
||||
if match:
|
||||
version = (int(match.group(1)), int(match.group(2)), int(match.group(3)))
|
||||
logger.debug(f"Detected rclone version: {version}")
|
||||
return version
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not determine rclone version: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def supports_create_empty_src_dirs() -> bool:
|
||||
"""Check if installed rclone supports --create-empty-src-dirs flag.
|
||||
|
||||
Returns:
|
||||
True if rclone version >= 1.64.0, False otherwise.
|
||||
"""
|
||||
version = get_rclone_version()
|
||||
if version is None:
|
||||
# If we can't determine version, assume older and skip the flag
|
||||
return False
|
||||
return version >= MIN_RCLONE_VERSION_EMPTY_DIRS
|
||||
|
||||
|
||||
@dataclass
|
||||
class SyncProject:
|
||||
"""Project configured for cloud sync.
|
||||
@@ -218,7 +260,6 @@ def project_bisync(
|
||||
"bisync",
|
||||
str(local_path),
|
||||
remote_path,
|
||||
"--create-empty-src-dirs",
|
||||
"--resilient",
|
||||
"--conflict-resolve=newer",
|
||||
"--max-delete=25",
|
||||
@@ -229,6 +270,10 @@ def project_bisync(
|
||||
str(state_path),
|
||||
]
|
||||
|
||||
# Add --create-empty-src-dirs if rclone version supports it (v1.64+)
|
||||
if supports_create_empty_src_dirs():
|
||||
cmd.append("--create-empty-src-dirs")
|
||||
|
||||
if verbose:
|
||||
cmd.append("--verbose")
|
||||
else:
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
"""utility functions for commands"""
|
||||
|
||||
from typing import Optional
|
||||
import asyncio
|
||||
from typing import Optional, TypeVar, Coroutine, Any
|
||||
|
||||
from mcp.server.fastmcp.exceptions import ToolError
|
||||
import typer
|
||||
|
||||
from rich.console import Console
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
|
||||
from basic_memory.mcp.tools.utils import call_post, call_get
|
||||
@@ -15,6 +17,30 @@ from basic_memory.schemas import ProjectInfoResponse
|
||||
|
||||
console = Console()
|
||||
|
||||
T = TypeVar("T")
|
||||
|
||||
|
||||
def run_with_cleanup(coro: Coroutine[Any, Any, T]) -> T:
|
||||
"""Run an async coroutine with proper database cleanup.
|
||||
|
||||
This helper ensures database connections are cleaned up before the event
|
||||
loop closes, preventing process hangs in CLI commands.
|
||||
|
||||
Args:
|
||||
coro: The coroutine to run
|
||||
|
||||
Returns:
|
||||
The result of the coroutine
|
||||
"""
|
||||
|
||||
async def _with_cleanup() -> T:
|
||||
try:
|
||||
return await coro
|
||||
finally:
|
||||
await db.shutdown_db()
|
||||
|
||||
return asyncio.run(_with_cleanup())
|
||||
|
||||
|
||||
async def run_sync(project: Optional[str] = None, force_full: bool = False):
|
||||
"""Run sync operation via API endpoint.
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
"""Format command for basic-memory CLI."""
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Optional
|
||||
|
||||
import typer
|
||||
from loguru import logger
|
||||
from rich.console import Console
|
||||
from rich.progress import Progress, SpinnerColumn, TextColumn
|
||||
|
||||
from basic_memory.cli.app import app
|
||||
from basic_memory.config import ConfigManager, get_project_config
|
||||
from basic_memory.file_utils import format_file
|
||||
|
||||
console = Console()
|
||||
|
||||
|
||||
def is_markdown_extension(path: Path) -> bool:
|
||||
"""Check if file has a markdown extension."""
|
||||
return path.suffix.lower() in (".md", ".markdown")
|
||||
|
||||
|
||||
async def format_single_file(file_path: Path, app_config) -> tuple[Path, bool, Optional[str]]:
|
||||
"""Format a single file.
|
||||
|
||||
Returns:
|
||||
Tuple of (path, success, error_message)
|
||||
"""
|
||||
try:
|
||||
result = await format_file(
|
||||
file_path, app_config, is_markdown=is_markdown_extension(file_path)
|
||||
)
|
||||
if result is not None:
|
||||
return (file_path, True, None)
|
||||
else:
|
||||
return (file_path, False, "No formatter configured or formatting skipped")
|
||||
except Exception as e:
|
||||
return (file_path, False, str(e))
|
||||
|
||||
|
||||
async def format_files(
|
||||
paths: list[Path], app_config, show_progress: bool = True
|
||||
) -> tuple[int, int, list[tuple[Path, str]]]:
|
||||
"""Format multiple files.
|
||||
|
||||
Returns:
|
||||
Tuple of (formatted_count, skipped_count, errors)
|
||||
"""
|
||||
formatted = 0
|
||||
skipped = 0
|
||||
errors: list[tuple[Path, str]] = []
|
||||
|
||||
if show_progress:
|
||||
with Progress(
|
||||
SpinnerColumn(),
|
||||
TextColumn("[progress.description]{task.description}"),
|
||||
console=console,
|
||||
) as progress:
|
||||
task = progress.add_task("Formatting files...", total=len(paths))
|
||||
|
||||
for file_path in paths:
|
||||
path, success, error = await format_single_file(file_path, app_config)
|
||||
if success:
|
||||
formatted += 1
|
||||
elif error and "No formatter configured" not in error:
|
||||
errors.append((path, error))
|
||||
else:
|
||||
skipped += 1
|
||||
progress.update(task, advance=1)
|
||||
else:
|
||||
for file_path in paths:
|
||||
path, success, error = await format_single_file(file_path, app_config)
|
||||
if success:
|
||||
formatted += 1
|
||||
elif error and "No formatter configured" not in error:
|
||||
errors.append((path, error))
|
||||
else:
|
||||
skipped += 1
|
||||
|
||||
return formatted, skipped, errors
|
||||
|
||||
|
||||
async def run_format(
|
||||
path: Optional[Path] = None,
|
||||
project: Optional[str] = None,
|
||||
) -> None:
|
||||
"""Run the format command."""
|
||||
app_config = ConfigManager().config
|
||||
|
||||
# Check if formatting is enabled
|
||||
if (
|
||||
not app_config.format_on_save
|
||||
and not app_config.formatter_command
|
||||
and not app_config.formatters
|
||||
):
|
||||
console.print(
|
||||
"[yellow]No formatters configured. Set format_on_save=true and "
|
||||
"formatter_command or formatters in your config.[/yellow]"
|
||||
)
|
||||
console.print(
|
||||
"\nExample config (~/.basic-memory/config.json):\n"
|
||||
' "format_on_save": true,\n'
|
||||
' "formatter_command": "prettier --write {file}"\n'
|
||||
)
|
||||
raise typer.Exit(1)
|
||||
|
||||
# Temporarily enable format_on_save for this command
|
||||
# (so format_file actually runs the formatter)
|
||||
original_format_on_save = app_config.format_on_save
|
||||
app_config.format_on_save = True
|
||||
|
||||
try:
|
||||
# Determine which files to format
|
||||
if path:
|
||||
# Format specific file or directory
|
||||
if path.is_file():
|
||||
files = [path]
|
||||
elif path.is_dir():
|
||||
# Find all markdown and json files
|
||||
files = (
|
||||
list(path.rglob("*.md"))
|
||||
+ list(path.rglob("*.json"))
|
||||
+ list(path.rglob("*.canvas"))
|
||||
)
|
||||
else:
|
||||
console.print(f"[red]Path not found: {path}[/red]")
|
||||
raise typer.Exit(1)
|
||||
else:
|
||||
# Format all files in project
|
||||
project_config = get_project_config(project)
|
||||
project_path = Path(project_config.home)
|
||||
|
||||
if not project_path.exists():
|
||||
console.print(f"[red]Project path not found: {project_path}[/red]")
|
||||
raise typer.Exit(1)
|
||||
|
||||
# Find all markdown and json files
|
||||
files = (
|
||||
list(project_path.rglob("*.md"))
|
||||
+ list(project_path.rglob("*.json"))
|
||||
+ list(project_path.rglob("*.canvas"))
|
||||
)
|
||||
|
||||
if not files:
|
||||
console.print("[yellow]No files found to format.[/yellow]")
|
||||
return
|
||||
|
||||
console.print(f"Found {len(files)} file(s) to format...")
|
||||
|
||||
formatted, skipped, errors = await format_files(files, app_config)
|
||||
|
||||
# Print summary
|
||||
console.print()
|
||||
if formatted > 0:
|
||||
console.print(f"[green]Formatted: {formatted} file(s)[/green]")
|
||||
if skipped > 0:
|
||||
console.print(f"[dim]Skipped: {skipped} file(s) (no formatter for extension)[/dim]")
|
||||
if errors:
|
||||
console.print(f"[red]Errors: {len(errors)} file(s)[/red]")
|
||||
for path, error in errors:
|
||||
console.print(f" [red]{path}[/red]: {error}")
|
||||
|
||||
finally:
|
||||
# Restore original setting
|
||||
app_config.format_on_save = original_format_on_save
|
||||
|
||||
|
||||
@app.command()
|
||||
def format(
|
||||
path: Annotated[
|
||||
Optional[Path],
|
||||
typer.Argument(help="File or directory to format. Defaults to current project."),
|
||||
] = None,
|
||||
project: Annotated[
|
||||
Optional[str],
|
||||
typer.Option("--project", "-p", help="Project name to format."),
|
||||
] = None,
|
||||
) -> None:
|
||||
"""Format files using configured formatters.
|
||||
|
||||
Uses the formatter_command or formatters settings from your config.
|
||||
By default, formats all .md, .json, and .canvas files in the current project.
|
||||
|
||||
Examples:
|
||||
basic-memory format # Format all files in current project
|
||||
basic-memory format --project research # Format files in specific project
|
||||
basic-memory format notes/meeting.md # Format a specific file
|
||||
basic-memory format notes/ # Format all files in directory
|
||||
"""
|
||||
try:
|
||||
asyncio.run(run_format(path, project))
|
||||
except Exception as e:
|
||||
if not isinstance(e, typer.Exit):
|
||||
logger.error(f"Error formatting files: {e}")
|
||||
console.print(f"[red]Error formatting files: {e}[/red]")
|
||||
raise typer.Exit(code=1)
|
||||
raise
|
||||
@@ -7,7 +7,7 @@ from typing import Annotated
|
||||
|
||||
import typer
|
||||
from basic_memory.cli.app import import_app
|
||||
from basic_memory.config import get_project_config
|
||||
from basic_memory.config import ConfigManager, get_project_config
|
||||
from basic_memory.importers import ChatGPTImporter
|
||||
from basic_memory.markdown import EntityParser, MarkdownProcessor
|
||||
from loguru import logger
|
||||
@@ -20,8 +20,9 @@ console = Console()
|
||||
async def get_markdown_processor() -> MarkdownProcessor:
|
||||
"""Get MarkdownProcessor instance."""
|
||||
config = get_project_config()
|
||||
app_config = ConfigManager().config
|
||||
entity_parser = EntityParser(config.home)
|
||||
return MarkdownProcessor(entity_parser)
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
@import_app.command(name="chatgpt", help="Import conversations from ChatGPT JSON export.")
|
||||
|
||||
@@ -7,7 +7,7 @@ from typing import Annotated
|
||||
|
||||
import typer
|
||||
from basic_memory.cli.app import claude_app
|
||||
from basic_memory.config import get_project_config
|
||||
from basic_memory.config import ConfigManager, get_project_config
|
||||
from basic_memory.importers.claude_conversations_importer import ClaudeConversationsImporter
|
||||
from basic_memory.markdown import EntityParser, MarkdownProcessor
|
||||
from loguru import logger
|
||||
@@ -20,8 +20,9 @@ console = Console()
|
||||
async def get_markdown_processor() -> MarkdownProcessor:
|
||||
"""Get MarkdownProcessor instance."""
|
||||
config = get_project_config()
|
||||
app_config = ConfigManager().config
|
||||
entity_parser = EntityParser(config.home)
|
||||
return MarkdownProcessor(entity_parser)
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
@claude_app.command(name="conversations", help="Import chat conversations from Claude.ai.")
|
||||
|
||||
@@ -7,7 +7,7 @@ from typing import Annotated
|
||||
|
||||
import typer
|
||||
from basic_memory.cli.app import claude_app
|
||||
from basic_memory.config import get_project_config
|
||||
from basic_memory.config import ConfigManager, get_project_config
|
||||
from basic_memory.importers.claude_projects_importer import ClaudeProjectsImporter
|
||||
from basic_memory.markdown import EntityParser, MarkdownProcessor
|
||||
from loguru import logger
|
||||
@@ -20,8 +20,9 @@ console = Console()
|
||||
async def get_markdown_processor() -> MarkdownProcessor:
|
||||
"""Get MarkdownProcessor instance."""
|
||||
config = get_project_config()
|
||||
app_config = ConfigManager().config
|
||||
entity_parser = EntityParser(config.home)
|
||||
return MarkdownProcessor(entity_parser)
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
@claude_app.command(name="projects", help="Import projects from Claude.ai.")
|
||||
|
||||
@@ -7,7 +7,7 @@ from typing import Annotated
|
||||
|
||||
import typer
|
||||
from basic_memory.cli.app import import_app
|
||||
from basic_memory.config import get_project_config
|
||||
from basic_memory.config import ConfigManager, get_project_config
|
||||
from basic_memory.importers.memory_json_importer import MemoryJsonImporter
|
||||
from basic_memory.markdown import EntityParser, MarkdownProcessor
|
||||
from loguru import logger
|
||||
@@ -20,8 +20,9 @@ console = Console()
|
||||
async def get_markdown_processor() -> MarkdownProcessor:
|
||||
"""Get MarkdownProcessor instance."""
|
||||
config = get_project_config()
|
||||
app_config = ConfigManager().config
|
||||
entity_parser = EntityParser(config.home)
|
||||
return MarkdownProcessor(entity_parser)
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
@import_app.command()
|
||||
|
||||
@@ -1,14 +1,13 @@
|
||||
"""MCP server command with streamable HTTP transport."""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import typer
|
||||
from typing import Optional
|
||||
|
||||
from basic_memory.cli.app import app
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.config import ConfigManager, init_mcp_logging
|
||||
|
||||
# Import mcp instance
|
||||
# Import mcp instance (has lifespan that handles initialization and file sync)
|
||||
from basic_memory.mcp.server import mcp as mcp_server # pragma: no cover
|
||||
|
||||
# Import mcp tools to register them
|
||||
@@ -17,8 +16,6 @@ import basic_memory.mcp.tools # noqa: F401 # pragma: no cover
|
||||
# Import prompts to register them
|
||||
import basic_memory.mcp.prompts # noqa: F401 # pragma: no cover
|
||||
from loguru import logger
|
||||
import threading
|
||||
from basic_memory.services.initialization import initialize_file_sync
|
||||
|
||||
config = ConfigManager().config
|
||||
|
||||
@@ -43,7 +40,11 @@ if not config.cloud_mode_enabled:
|
||||
- stdio: Standard I/O (good for local usage)
|
||||
- streamable-http: Recommended for web deployments (default)
|
||||
- sse: Server-Sent Events (for compatibility with existing clients)
|
||||
|
||||
Initialization, file sync, and cleanup are handled by the MCP server's lifespan.
|
||||
"""
|
||||
# Initialize logging for MCP (file only, stdout breaks protocol)
|
||||
init_mcp_logging()
|
||||
|
||||
# Validate and set project constraint if specified
|
||||
if project:
|
||||
@@ -57,27 +58,8 @@ if not config.cloud_mode_enabled:
|
||||
os.environ["BASIC_MEMORY_MCP_PROJECT"] = project_name
|
||||
logger.info(f"MCP server constrained to project: {project_name}")
|
||||
|
||||
app_config = ConfigManager().config
|
||||
|
||||
def run_file_sync():
|
||||
"""Run file sync in a separate thread with its own event loop."""
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
try:
|
||||
loop.run_until_complete(initialize_file_sync(app_config))
|
||||
except Exception as e:
|
||||
logger.error(f"File sync error: {e}", err=True)
|
||||
finally:
|
||||
loop.close()
|
||||
|
||||
logger.info(f"Sync changes enabled: {app_config.sync_changes}")
|
||||
if app_config.sync_changes:
|
||||
# Start the sync thread
|
||||
sync_thread = threading.Thread(target=run_file_sync, daemon=True)
|
||||
sync_thread.start()
|
||||
logger.info("Started file sync in background")
|
||||
|
||||
# Now run the MCP server (blocks)
|
||||
# Run the MCP server (blocks)
|
||||
# Lifespan handles: initialization, migrations, file sync, cleanup
|
||||
logger.info(f"Starting MCP server with {transport.upper()} transport")
|
||||
|
||||
if transport == "stdio":
|
||||
|
||||
@@ -16,14 +16,9 @@ from datetime import datetime
|
||||
|
||||
from rich.panel import Panel
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.schemas.project_info import ProjectList
|
||||
from basic_memory.mcp.tools.utils import call_post
|
||||
from basic_memory.schemas.project_info import ProjectStatusResponse
|
||||
from basic_memory.mcp.tools.utils import call_delete
|
||||
from basic_memory.mcp.tools.utils import call_put
|
||||
from basic_memory.mcp.tools.utils import call_get, call_post, call_delete, call_put, call_patch
|
||||
from basic_memory.schemas.project_info import ProjectList, ProjectStatusResponse
|
||||
from basic_memory.utils import generate_permalink, normalize_project_path
|
||||
from basic_memory.mcp.tools.utils import call_patch
|
||||
|
||||
# Import rclone commands for project sync
|
||||
from basic_memory.cli.commands.cloud.rclone_commands import (
|
||||
@@ -254,9 +249,17 @@ def remove_project(
|
||||
|
||||
async def _remove_project():
|
||||
async with get_client() as client:
|
||||
# Convert name to permalink for efficient resolution
|
||||
project_permalink = generate_permalink(name)
|
||||
|
||||
# Use v2 project resolver to find project ID by permalink
|
||||
resolve_data = {"identifier": project_permalink}
|
||||
response = await call_post(client, "/v2/projects/resolve", json=resolve_data)
|
||||
target_project = response.json()
|
||||
|
||||
# Use v2 API with project ID
|
||||
response = await call_delete(
|
||||
client, f"/projects/{project_permalink}?delete_notes={delete_notes}"
|
||||
client, f"/v2/projects/{target_project['project_id']}?delete_notes={delete_notes}"
|
||||
)
|
||||
return ProjectStatusResponse.model_validate(response.json())
|
||||
|
||||
@@ -329,8 +332,18 @@ def set_default_project(
|
||||
|
||||
async def _set_default():
|
||||
async with get_client() as client:
|
||||
# Convert name to permalink for efficient resolution
|
||||
project_permalink = generate_permalink(name)
|
||||
response = await call_put(client, f"/projects/{project_permalink}/default")
|
||||
|
||||
# Use v2 project resolver to find project ID by permalink
|
||||
resolve_data = {"identifier": project_permalink}
|
||||
response = await call_post(client, "/v2/projects/resolve", json=resolve_data)
|
||||
target_project = response.json()
|
||||
|
||||
# Use v2 API with project ID
|
||||
response = await call_put(
|
||||
client, f"/v2/projects/{target_project['project_id']}/default"
|
||||
)
|
||||
return ProjectStatusResponse.model_validate(response.json())
|
||||
|
||||
try:
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
"""Status command for basic-memory CLI."""
|
||||
|
||||
import asyncio
|
||||
from typing import Set, Dict
|
||||
from typing import Annotated, Optional
|
||||
|
||||
@@ -165,8 +164,10 @@ def status(
|
||||
verbose: bool = typer.Option(False, "--verbose", "-v", help="Show detailed file information"),
|
||||
):
|
||||
"""Show sync status between files and database."""
|
||||
from basic_memory.cli.commands.command_utils import run_with_cleanup
|
||||
|
||||
try:
|
||||
asyncio.run(run_status(project, verbose)) # pragma: no cover
|
||||
run_with_cleanup(run_status(project, verbose)) # pragma: no cover
|
||||
except Exception as e:
|
||||
logger.error(f"Error checking status: {e}")
|
||||
typer.echo(f"Error checking status: {e}", err=True)
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Telemetry commands for basic-memory CLI."""
|
||||
|
||||
import typer
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
|
||||
from basic_memory.cli.app import app
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
console = Console()
|
||||
|
||||
# Create telemetry subcommand group
|
||||
telemetry_app = typer.Typer(help="Manage anonymous telemetry settings")
|
||||
app.add_typer(telemetry_app, name="telemetry")
|
||||
|
||||
|
||||
@telemetry_app.command("enable")
|
||||
def enable() -> None:
|
||||
"""Enable anonymous telemetry.
|
||||
|
||||
Telemetry helps improve Basic Memory by collecting anonymous usage data.
|
||||
No personal data, note content, or file paths are ever collected.
|
||||
"""
|
||||
config_manager = ConfigManager()
|
||||
config = config_manager.config
|
||||
config.telemetry_enabled = True
|
||||
config_manager.save_config(config)
|
||||
console.print("[green]Telemetry enabled[/green]")
|
||||
console.print("[dim]Thank you for helping improve Basic Memory![/dim]")
|
||||
|
||||
|
||||
@telemetry_app.command("disable")
|
||||
def disable() -> None:
|
||||
"""Disable anonymous telemetry.
|
||||
|
||||
You can re-enable telemetry anytime with: bm telemetry enable
|
||||
"""
|
||||
config_manager = ConfigManager()
|
||||
config = config_manager.config
|
||||
config.telemetry_enabled = False
|
||||
config_manager.save_config(config)
|
||||
console.print("[yellow]Telemetry disabled[/yellow]")
|
||||
|
||||
|
||||
@telemetry_app.command("status")
|
||||
def status() -> None:
|
||||
"""Show current telemetry status and what's collected."""
|
||||
from basic_memory.telemetry import get_install_id, TELEMETRY_DOCS_URL
|
||||
|
||||
config = ConfigManager().config
|
||||
|
||||
status_text = (
|
||||
"[green]enabled[/green]" if config.telemetry_enabled else "[yellow]disabled[/yellow]"
|
||||
)
|
||||
|
||||
console.print(f"\nTelemetry: {status_text}")
|
||||
console.print(f"Install ID: [dim]{get_install_id()}[/dim]")
|
||||
console.print()
|
||||
|
||||
what_we_collect = """
|
||||
[bold]What we collect:[/bold]
|
||||
- App version, Python version, OS, architecture
|
||||
- Feature usage (which MCP tools and CLI commands)
|
||||
- Sync statistics (entity count, duration)
|
||||
- Error types (sanitized, no file paths)
|
||||
|
||||
[bold]What we NEVER collect:[/bold]
|
||||
- Note content, file names, or paths
|
||||
- Personal information
|
||||
- IP addresses
|
||||
"""
|
||||
|
||||
console.print(
|
||||
Panel(
|
||||
what_we_collect.strip(),
|
||||
title="Telemetry Details",
|
||||
border_style="blue",
|
||||
expand=False,
|
||||
)
|
||||
)
|
||||
console.print(f"[dim]Details: {TELEMETRY_DOCS_URL}[/dim]")
|
||||
@@ -13,9 +13,16 @@ from basic_memory.cli.commands import ( # noqa: F401 # pragma: no cover
|
||||
mcp,
|
||||
project,
|
||||
status,
|
||||
telemetry,
|
||||
tool,
|
||||
)
|
||||
|
||||
# Re-apply warning filter AFTER all imports
|
||||
# (authlib adds a DeprecationWarning filter that overrides ours)
|
||||
import warnings # pragma: no cover
|
||||
|
||||
warnings.filterwarnings("ignore") # pragma: no cover
|
||||
|
||||
if __name__ == "__main__": # pragma: no cover
|
||||
# start the app
|
||||
app()
|
||||
|
||||
+149
-71
@@ -9,10 +9,9 @@ from typing import Any, Dict, Literal, Optional, List, Tuple
|
||||
from enum import Enum
|
||||
|
||||
from loguru import logger
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from pydantic import BaseModel, Field, model_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
import basic_memory
|
||||
from basic_memory.utils import setup_logging, generate_permalink
|
||||
|
||||
|
||||
@@ -100,13 +99,32 @@ class BasicMemoryConfig(BaseSettings):
|
||||
description="Database connection URL. For Postgres, use postgresql+asyncpg://user:pass@host:port/db. If not set, SQLite will use default path.",
|
||||
)
|
||||
|
||||
# Database connection pool configuration (Postgres only)
|
||||
db_pool_size: int = Field(
|
||||
default=20,
|
||||
description="Number of connections to keep in the pool (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
db_pool_overflow: int = Field(
|
||||
default=40,
|
||||
description="Max additional connections beyond pool_size under load (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
db_pool_recycle: int = Field(
|
||||
default=180,
|
||||
description="Recycle connections after N seconds to prevent stale connections. Default 180s works well with Neon's ~5 minute scale-to-zero (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
|
||||
# Watch service configuration
|
||||
sync_delay: int = Field(
|
||||
default=1000, description="Milliseconds to wait after changes before syncing", gt=0
|
||||
)
|
||||
|
||||
watch_project_reload_interval: int = Field(
|
||||
default=30, description="Seconds between reloading project list in watch service", gt=0
|
||||
default=300,
|
||||
description="Seconds between reloading project list in watch service. Higher values reduce CPU usage by minimizing watcher restarts. Default 300s (5 min) balances efficiency with responsiveness to new projects.",
|
||||
gt=0,
|
||||
)
|
||||
|
||||
# update permalinks on move
|
||||
@@ -147,6 +165,28 @@ class BasicMemoryConfig(BaseSettings):
|
||||
description="Skip expensive initialization synchronization. Useful for cloud/stateless deployments where project reconciliation is not needed.",
|
||||
)
|
||||
|
||||
# File formatting configuration
|
||||
format_on_save: bool = Field(
|
||||
default=False,
|
||||
description="Automatically format files after saving using configured formatter. Disabled by default.",
|
||||
)
|
||||
|
||||
formatter_command: Optional[str] = Field(
|
||||
default=None,
|
||||
description="External formatter command. Use {file} as placeholder for file path. If not set, uses built-in mdformat (Python, no Node.js required). Set to 'npx prettier --write {file}' for Prettier.",
|
||||
)
|
||||
|
||||
formatters: Dict[str, str] = Field(
|
||||
default_factory=dict,
|
||||
description="Per-extension formatters. Keys are extensions (without dot), values are commands. Example: {'md': 'prettier --write {file}', 'json': 'prettier --write {file}'}",
|
||||
)
|
||||
|
||||
formatter_timeout: float = Field(
|
||||
default=5.0,
|
||||
description="Maximum seconds to wait for formatter to complete",
|
||||
gt=0,
|
||||
)
|
||||
|
||||
# Project path constraints
|
||||
project_root: Optional[str] = Field(
|
||||
default=None,
|
||||
@@ -181,6 +221,34 @@ class BasicMemoryConfig(BaseSettings):
|
||||
description="Cloud project sync configuration mapping project names to their local paths and sync state",
|
||||
)
|
||||
|
||||
# Telemetry configuration (Homebrew-style opt-out)
|
||||
telemetry_enabled: bool = Field(
|
||||
default=True,
|
||||
description="Send anonymous usage statistics to help improve Basic Memory. Disable with: bm telemetry disable",
|
||||
)
|
||||
|
||||
telemetry_notice_shown: bool = Field(
|
||||
default=False,
|
||||
description="Whether the one-time telemetry notice has been shown to the user",
|
||||
)
|
||||
|
||||
@property
|
||||
def is_test_env(self) -> bool:
|
||||
"""Check if running in a test environment.
|
||||
|
||||
Returns True if any of:
|
||||
- env field is set to "test"
|
||||
- BASIC_MEMORY_ENV environment variable is "test"
|
||||
- PYTEST_CURRENT_TEST environment variable is set (pytest is running)
|
||||
|
||||
Used to disable features like telemetry and file watchers during tests.
|
||||
"""
|
||||
return (
|
||||
self.env == "test"
|
||||
or os.getenv("BASIC_MEMORY_ENV", "").lower() == "test"
|
||||
or os.getenv("PYTEST_CURRENT_TEST") is not None
|
||||
)
|
||||
|
||||
@property
|
||||
def cloud_mode_enabled(self) -> bool:
|
||||
"""Check if cloud mode is enabled.
|
||||
@@ -197,6 +265,36 @@ class BasicMemoryConfig(BaseSettings):
|
||||
# Fall back to config file value
|
||||
return self.cloud_mode
|
||||
|
||||
@classmethod
|
||||
def for_cloud_tenant(
|
||||
cls,
|
||||
database_url: str,
|
||||
projects: Optional[Dict[str, str]] = None,
|
||||
) -> "BasicMemoryConfig":
|
||||
"""Create config for cloud tenant - no config.json, database is source of truth.
|
||||
|
||||
This factory method creates a BasicMemoryConfig suitable for cloud deployments
|
||||
where:
|
||||
- Database is Postgres (Neon), not SQLite
|
||||
- Projects are discovered from the database, not config file
|
||||
- Path validation is skipped (no local filesystem in cloud)
|
||||
- Initialization sync is skipped (stateless deployment)
|
||||
|
||||
Args:
|
||||
database_url: Postgres connection URL for tenant database
|
||||
projects: Optional project mapping (usually empty, discovered from DB)
|
||||
|
||||
Returns:
|
||||
BasicMemoryConfig configured for cloud mode
|
||||
"""
|
||||
return cls(
|
||||
database_backend=DatabaseBackend.POSTGRES,
|
||||
database_url=database_url,
|
||||
projects=projects or {},
|
||||
cloud_mode=True,
|
||||
skip_initialization_sync=True,
|
||||
)
|
||||
|
||||
model_config = SettingsConfigDict(
|
||||
env_prefix="BASIC_MEMORY_",
|
||||
extra="ignore",
|
||||
@@ -213,6 +311,10 @@ class BasicMemoryConfig(BaseSettings):
|
||||
|
||||
def model_post_init(self, __context: Any) -> None:
|
||||
"""Ensure configuration is valid after initialization."""
|
||||
# Skip project initialization in cloud mode - projects are discovered from DB
|
||||
if self.database_backend == DatabaseBackend.POSTGRES:
|
||||
return
|
||||
|
||||
# Ensure at least one project exists; if none exist then create main
|
||||
if not self.projects: # pragma: no cover
|
||||
self.projects["main"] = str(
|
||||
@@ -255,19 +357,26 @@ class BasicMemoryConfig(BaseSettings):
|
||||
"""Get all configured projects as ProjectConfig objects."""
|
||||
return [ProjectConfig(name=name, home=Path(path)) for name, path in self.projects.items()]
|
||||
|
||||
@field_validator("projects")
|
||||
@classmethod
|
||||
def ensure_project_paths_exists(cls, v: Dict[str, str]) -> Dict[str, str]: # pragma: no cover
|
||||
"""Ensure project path exists."""
|
||||
for name, path_value in v.items():
|
||||
@model_validator(mode="after")
|
||||
def ensure_project_paths_exists(self) -> "BasicMemoryConfig": # pragma: no cover
|
||||
"""Ensure project paths exist.
|
||||
|
||||
Skips path creation when using Postgres backend (cloud mode) since
|
||||
cloud tenants don't use local filesystem paths.
|
||||
"""
|
||||
# Skip path creation for cloud mode - no local filesystem
|
||||
if self.database_backend == DatabaseBackend.POSTGRES:
|
||||
return self
|
||||
|
||||
for name, path_value in self.projects.items():
|
||||
path = Path(path_value)
|
||||
if not Path(path).exists():
|
||||
if not path.exists():
|
||||
try:
|
||||
path.mkdir(parents=True)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create project path: {e}")
|
||||
raise e
|
||||
return v
|
||||
return self
|
||||
|
||||
@property
|
||||
def data_dir_path(self):
|
||||
@@ -470,69 +579,38 @@ def save_basic_memory_config(file_path: Path, config: BasicMemoryConfig) -> None
|
||||
logger.error(f"Failed to save config: {e}")
|
||||
|
||||
|
||||
# setup logging to a single log file in user home directory
|
||||
user_home = Path.home()
|
||||
log_dir = user_home / DATA_DIR_NAME
|
||||
log_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Logging initialization functions for different entry points
|
||||
|
||||
|
||||
# Process info for logging
|
||||
def get_process_name(): # pragma: no cover
|
||||
def init_cli_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for CLI commands - file only.
|
||||
|
||||
CLI commands should not log to stdout to avoid interfering with
|
||||
command output and shell integration.
|
||||
"""
|
||||
get the type of process for logging
|
||||
"""
|
||||
import sys
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
if "sync" in sys.argv:
|
||||
return "sync"
|
||||
elif "mcp" in sys.argv:
|
||||
return "mcp"
|
||||
elif "cli" in sys.argv:
|
||||
return "cli"
|
||||
|
||||
def init_mcp_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for MCP server - file only.
|
||||
|
||||
MCP server must not log to stdout as it would corrupt the
|
||||
JSON-RPC protocol communication.
|
||||
"""
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
|
||||
def init_api_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for API server.
|
||||
|
||||
Cloud mode (BASIC_MEMORY_CLOUD_MODE=1): stdout with structured context
|
||||
Local mode: file only
|
||||
"""
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
cloud_mode = os.getenv("BASIC_MEMORY_CLOUD_MODE", "").lower() in ("1", "true")
|
||||
if cloud_mode:
|
||||
setup_logging(log_level=log_level, log_to_stdout=True, structured_context=True)
|
||||
else:
|
||||
return "api"
|
||||
|
||||
|
||||
process_name = get_process_name()
|
||||
|
||||
# Global flag to track if logging has been set up
|
||||
_LOGGING_SETUP = False
|
||||
|
||||
|
||||
# Logging
|
||||
|
||||
|
||||
def setup_basic_memory_logging(): # pragma: no cover
|
||||
"""Set up logging for basic-memory, ensuring it only happens once."""
|
||||
global _LOGGING_SETUP
|
||||
if _LOGGING_SETUP:
|
||||
# We can't log before logging is set up
|
||||
# print("Skipping duplicate logging setup")
|
||||
return
|
||||
|
||||
# Check for console logging environment variable - accept more truthy values
|
||||
console_logging_env = os.getenv("BASIC_MEMORY_CONSOLE_LOGGING", "false").lower()
|
||||
console_logging = console_logging_env in ("true", "1", "yes", "on")
|
||||
|
||||
# Check for log level environment variable first, fall back to config
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL")
|
||||
if not log_level:
|
||||
config_manager = ConfigManager()
|
||||
log_level = config_manager.config.log_level
|
||||
|
||||
config_manager = ConfigManager()
|
||||
config = get_project_config()
|
||||
setup_logging(
|
||||
env=config_manager.config.env,
|
||||
home_dir=user_home, # Use user home for logs
|
||||
log_level=log_level,
|
||||
log_file=f"{DATA_DIR_NAME}/basic-memory-{process_name}.log",
|
||||
console=console_logging,
|
||||
)
|
||||
|
||||
logger.info(f"Basic Memory {basic_memory.__version__} (Project: {config.project})")
|
||||
_LOGGING_SETUP = True
|
||||
|
||||
|
||||
# Set up logging
|
||||
setup_basic_memory_logging()
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
+36
-4
@@ -1,5 +1,6 @@
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
from contextlib import asynccontextmanager
|
||||
from enum import Enum, auto
|
||||
from pathlib import Path
|
||||
@@ -23,6 +24,21 @@ from sqlalchemy.pool import NullPool
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.sqlite_search_repository import SQLiteSearchRepository
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Windows event loop policy
|
||||
# -----------------------------------------------------------------------------
|
||||
# On Windows, the default ProactorEventLoop has known rough edges with aiosqlite
|
||||
# during shutdown/teardown (threads posting results to a loop that's closing),
|
||||
# which can manifest as:
|
||||
# - "RuntimeError: Event loop is closed"
|
||||
# - "IndexError: pop from an empty deque"
|
||||
#
|
||||
# The SelectorEventLoop doesn't support subprocess operations, so code that uses
|
||||
# asyncio.create_subprocess_shell() (like sync_service._quick_count_files) must
|
||||
# detect Windows and use fallback implementations.
|
||||
if sys.platform == "win32": # pragma: no cover
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
|
||||
# Module level state
|
||||
_engine: Optional[AsyncEngine] = None
|
||||
_session_maker: Optional[async_sessionmaker[AsyncSession]] = None
|
||||
@@ -190,21 +206,37 @@ def _create_sqlite_engine(db_url: str, db_type: DatabaseType) -> AsyncEngine:
|
||||
return engine
|
||||
|
||||
|
||||
def _create_postgres_engine(db_url: str) -> AsyncEngine:
|
||||
def _create_postgres_engine(db_url: str, config: BasicMemoryConfig) -> AsyncEngine:
|
||||
"""Create Postgres async engine with appropriate configuration.
|
||||
|
||||
Args:
|
||||
db_url: Postgres connection URL (postgresql+asyncpg://...)
|
||||
config: BasicMemoryConfig with pool settings
|
||||
|
||||
Returns:
|
||||
Configured async engine for Postgres
|
||||
"""
|
||||
# Postgres with asyncpg - use standard async connection
|
||||
# Use NullPool connection issues.
|
||||
# Assume connection pooler like PgBouncer handles connection pooling.
|
||||
engine = create_async_engine(
|
||||
db_url,
|
||||
echo=False,
|
||||
pool_pre_ping=True, # Verify connections before using them
|
||||
poolclass=NullPool, # No pooling - fresh connection per request
|
||||
connect_args={
|
||||
# Disable statement cache to avoid issues with prepared statements on reconnect
|
||||
"statement_cache_size": 0,
|
||||
# Allow 30s for commands (Neon cold start can take 2-5s, sometimes longer)
|
||||
"command_timeout": 30,
|
||||
# Allow 30s for initial connection (Neon wake-up time)
|
||||
"timeout": 30,
|
||||
"server_settings": {
|
||||
"application_name": "basic-memory",
|
||||
# Statement timeout for queries (30s to allow for cold start)
|
||||
"statement_timeout": "30s",
|
||||
},
|
||||
},
|
||||
)
|
||||
logger.debug("Created Postgres engine with NullPool (no connection pooling)")
|
||||
|
||||
return engine
|
||||
|
||||
@@ -228,7 +260,7 @@ def _create_engine_and_session(
|
||||
# Delegate to backend-specific engine creation
|
||||
# Check explicit POSTGRES type first, then config setting
|
||||
if db_type == DatabaseType.POSTGRES or config.database_backend == DatabaseBackend.POSTGRES:
|
||||
engine = _create_postgres_engine(db_url)
|
||||
engine = _create_postgres_engine(db_url, config)
|
||||
else:
|
||||
engine = _create_sqlite_engine(db_url, db_type)
|
||||
|
||||
|
||||
+22
-12
@@ -351,28 +351,33 @@ async def get_entity_parser_v2(project_config: ProjectConfigV2Dep) -> EntityPars
|
||||
EntityParserV2Dep = Annotated["EntityParser", Depends(get_entity_parser_v2)]
|
||||
|
||||
|
||||
async def get_markdown_processor(entity_parser: EntityParserDep) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser)
|
||||
async def get_markdown_processor(
|
||||
entity_parser: EntityParserDep, app_config: AppConfigDep
|
||||
) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
MarkdownProcessorDep = Annotated[MarkdownProcessor, Depends(get_markdown_processor)]
|
||||
|
||||
|
||||
async def get_markdown_processor_v2(entity_parser: EntityParserV2Dep) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser)
|
||||
async def get_markdown_processor_v2(
|
||||
entity_parser: EntityParserV2Dep, app_config: AppConfigDep
|
||||
) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
|
||||
|
||||
MarkdownProcessorV2Dep = Annotated[MarkdownProcessor, Depends(get_markdown_processor_v2)]
|
||||
|
||||
|
||||
async def get_file_service(
|
||||
project_config: ProjectConfigDep, markdown_processor: MarkdownProcessorDep
|
||||
project_config: ProjectConfigDep,
|
||||
markdown_processor: MarkdownProcessorDep,
|
||||
app_config: AppConfigDep,
|
||||
) -> FileService:
|
||||
file_service = FileService(project_config.home, markdown_processor, app_config=app_config)
|
||||
logger.debug(
|
||||
f"Creating FileService for project: {project_config.name}, base_path: {project_config.home}"
|
||||
f"Created FileService for project: {project_config.name}, base_path: {project_config.home} "
|
||||
)
|
||||
file_service = FileService(project_config.home, markdown_processor)
|
||||
logger.debug(f"Created FileService for project: {file_service} ")
|
||||
return file_service
|
||||
|
||||
|
||||
@@ -380,13 +385,14 @@ FileServiceDep = Annotated[FileService, Depends(get_file_service)]
|
||||
|
||||
|
||||
async def get_file_service_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
project_config: ProjectConfigV2Dep,
|
||||
markdown_processor: MarkdownProcessorV2Dep,
|
||||
app_config: AppConfigDep,
|
||||
) -> FileService:
|
||||
file_service = FileService(project_config.home, markdown_processor, app_config=app_config)
|
||||
logger.debug(
|
||||
f"Creating FileService for project: {project_config.name}, base_path: {project_config.home}"
|
||||
f"Created FileService for project: {project_config.name}, base_path: {project_config.home}"
|
||||
)
|
||||
file_service = FileService(project_config.home, markdown_processor)
|
||||
logger.debug(f"Created FileService for project: {file_service} ")
|
||||
return file_service
|
||||
|
||||
|
||||
@@ -400,6 +406,7 @@ async def get_entity_service(
|
||||
entity_parser: EntityParserDep,
|
||||
file_service: FileServiceDep,
|
||||
link_resolver: "LinkResolverDep",
|
||||
search_service: "SearchServiceDep",
|
||||
app_config: AppConfigDep,
|
||||
) -> EntityService:
|
||||
"""Create EntityService with repository."""
|
||||
@@ -410,6 +417,7 @@ async def get_entity_service(
|
||||
entity_parser=entity_parser,
|
||||
file_service=file_service,
|
||||
link_resolver=link_resolver,
|
||||
search_service=search_service,
|
||||
app_config=app_config,
|
||||
)
|
||||
|
||||
@@ -424,6 +432,7 @@ async def get_entity_service_v2(
|
||||
entity_parser: EntityParserV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
link_resolver: "LinkResolverV2Dep",
|
||||
search_service: "SearchServiceV2Dep",
|
||||
app_config: AppConfigDep,
|
||||
) -> EntityService:
|
||||
"""Create EntityService for v2 API."""
|
||||
@@ -434,6 +443,7 @@ async def get_entity_service_v2(
|
||||
entity_parser=entity_parser,
|
||||
file_service=file_service,
|
||||
link_resolver=link_resolver,
|
||||
search_service=search_service,
|
||||
app_config=app_config,
|
||||
)
|
||||
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
"""Utilities for file operations."""
|
||||
|
||||
import asyncio
|
||||
import hashlib
|
||||
import shlex
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any, Dict, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, Optional, Union
|
||||
|
||||
import aiofiles
|
||||
import yaml
|
||||
@@ -12,6 +16,23 @@ from loguru import logger
|
||||
|
||||
from basic_memory.utils import FilePath
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from basic_memory.config import BasicMemoryConfig
|
||||
|
||||
|
||||
@dataclass
|
||||
class FileMetadata:
|
||||
"""File metadata for cloud-compatible file operations.
|
||||
|
||||
This dataclass provides a cloud-agnostic way to represent file metadata,
|
||||
enabling S3FileService to return metadata from head_object responses
|
||||
instead of mock stat_result with zeros.
|
||||
"""
|
||||
|
||||
size: int
|
||||
created_at: datetime
|
||||
modified_at: datetime
|
||||
|
||||
|
||||
class FileError(Exception):
|
||||
"""Base exception for file operations."""
|
||||
@@ -53,6 +74,28 @@ async def compute_checksum(content: Union[str, bytes]) -> str:
|
||||
raise FileError(f"Failed to compute checksum: {e}")
|
||||
|
||||
|
||||
# UTF-8 BOM character that can appear at the start of files
|
||||
UTF8_BOM = "\ufeff"
|
||||
|
||||
|
||||
def strip_bom(content: str) -> str:
|
||||
"""Strip UTF-8 BOM from the start of content if present.
|
||||
|
||||
BOM (Byte Order Mark) characters can be present in files created on Windows
|
||||
or copied from certain sources. They should be stripped before processing
|
||||
frontmatter. See issue #452.
|
||||
|
||||
Args:
|
||||
content: Content that may start with BOM
|
||||
|
||||
Returns:
|
||||
Content with BOM removed if present
|
||||
"""
|
||||
if content and content.startswith(UTF8_BOM):
|
||||
return content[1:]
|
||||
return content
|
||||
|
||||
|
||||
async def write_file_atomic(path: FilePath, content: str) -> None:
|
||||
"""
|
||||
Write file with atomic operation using temporary file.
|
||||
@@ -84,6 +127,168 @@ async def write_file_atomic(path: FilePath, content: str) -> None:
|
||||
raise FileWriteError(f"Failed to write file {path}: {e}")
|
||||
|
||||
|
||||
async def format_markdown_builtin(path: Path) -> Optional[str]:
|
||||
"""
|
||||
Format a markdown file using the built-in mdformat formatter.
|
||||
|
||||
Uses mdformat with GFM (GitHub Flavored Markdown) support for consistent
|
||||
formatting without requiring Node.js or external tools.
|
||||
|
||||
Args:
|
||||
path: Path to the markdown file to format
|
||||
|
||||
Returns:
|
||||
Formatted content if successful, None if formatting failed.
|
||||
"""
|
||||
try:
|
||||
import mdformat
|
||||
except ImportError:
|
||||
logger.warning(
|
||||
"mdformat not installed, skipping built-in formatting",
|
||||
path=str(path),
|
||||
)
|
||||
return None
|
||||
|
||||
try:
|
||||
# Read original content
|
||||
async with aiofiles.open(path, mode="r", encoding="utf-8") as f:
|
||||
content = await f.read()
|
||||
|
||||
# Format using mdformat with GFM and frontmatter extensions
|
||||
# mdformat is synchronous, so we run it in a thread executor
|
||||
loop = asyncio.get_event_loop()
|
||||
formatted_content = await loop.run_in_executor(
|
||||
None,
|
||||
lambda: mdformat.text(
|
||||
content,
|
||||
extensions={"gfm", "frontmatter"}, # GFM + YAML frontmatter support
|
||||
options={"wrap": "no"}, # Don't wrap lines
|
||||
),
|
||||
)
|
||||
|
||||
# Only write if content changed
|
||||
if formatted_content != content:
|
||||
async with aiofiles.open(path, mode="w", encoding="utf-8") as f:
|
||||
await f.write(formatted_content)
|
||||
|
||||
logger.debug(
|
||||
"Formatted file with mdformat",
|
||||
path=str(path),
|
||||
changed=formatted_content != content,
|
||||
)
|
||||
return formatted_content
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"mdformat formatting failed",
|
||||
path=str(path),
|
||||
error=str(e),
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
async def format_file(
|
||||
path: Path,
|
||||
config: "BasicMemoryConfig",
|
||||
is_markdown: bool = False,
|
||||
) -> Optional[str]:
|
||||
"""
|
||||
Format a file using configured formatter.
|
||||
|
||||
By default, uses the built-in mdformat formatter for markdown files (pure Python,
|
||||
no Node.js required). External formatters like Prettier can be configured via
|
||||
formatter_command or per-extension formatters.
|
||||
|
||||
Args:
|
||||
path: File to format
|
||||
config: Configuration with formatter settings
|
||||
is_markdown: Whether this is a markdown file (caller should use FileService.is_markdown)
|
||||
|
||||
Returns:
|
||||
Formatted content if successful, None if formatting was skipped or failed.
|
||||
Failures are logged as warnings but don't raise exceptions.
|
||||
"""
|
||||
if not config.format_on_save:
|
||||
return None
|
||||
|
||||
extension = path.suffix.lstrip(".")
|
||||
formatter = config.formatters.get(extension) or config.formatter_command
|
||||
|
||||
# Use built-in mdformat for markdown files when no external formatter configured
|
||||
if not formatter:
|
||||
if is_markdown:
|
||||
return await format_markdown_builtin(path)
|
||||
else:
|
||||
logger.debug("No formatter configured for extension", extension=extension)
|
||||
return None
|
||||
|
||||
# Use external formatter
|
||||
# Replace {file} placeholder with the actual path
|
||||
cmd = formatter.replace("{file}", str(path))
|
||||
|
||||
try:
|
||||
# Parse command into args list for safer execution (no shell=True)
|
||||
args = shlex.split(cmd)
|
||||
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
*args,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
|
||||
try:
|
||||
stdout, stderr = await asyncio.wait_for(
|
||||
proc.communicate(),
|
||||
timeout=config.formatter_timeout,
|
||||
)
|
||||
except asyncio.TimeoutError:
|
||||
proc.kill()
|
||||
await proc.wait()
|
||||
logger.warning(
|
||||
"Formatter timed out",
|
||||
path=str(path),
|
||||
timeout=config.formatter_timeout,
|
||||
)
|
||||
return None
|
||||
|
||||
if proc.returncode != 0:
|
||||
logger.warning(
|
||||
"Formatter exited with non-zero status",
|
||||
path=str(path),
|
||||
returncode=proc.returncode,
|
||||
stderr=stderr.decode("utf-8", errors="replace") if stderr else "",
|
||||
)
|
||||
# Still try to read the file - formatter may have partially worked
|
||||
# or the file may be unchanged
|
||||
|
||||
# Read formatted content
|
||||
async with aiofiles.open(path, mode="r", encoding="utf-8") as f:
|
||||
formatted_content = await f.read()
|
||||
|
||||
logger.debug(
|
||||
"Formatted file successfully",
|
||||
path=str(path),
|
||||
formatter=args[0] if args else formatter,
|
||||
)
|
||||
return formatted_content
|
||||
|
||||
except FileNotFoundError:
|
||||
# Formatter executable not found
|
||||
logger.warning(
|
||||
"Formatter executable not found",
|
||||
command=cmd.split()[0] if cmd else "",
|
||||
path=str(path),
|
||||
)
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"Formatter failed",
|
||||
path=str(path),
|
||||
error=str(e),
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def has_frontmatter(content: str) -> bool:
|
||||
"""
|
||||
Check if content contains valid YAML frontmatter.
|
||||
@@ -97,7 +302,8 @@ def has_frontmatter(content: str) -> bool:
|
||||
if not content:
|
||||
return False
|
||||
|
||||
content = content.strip()
|
||||
# Strip BOM before checking for frontmatter markers
|
||||
content = strip_bom(content).strip()
|
||||
if not content.startswith("---"):
|
||||
return False
|
||||
|
||||
@@ -118,6 +324,8 @@ def parse_frontmatter(content: str) -> Dict[str, Any]:
|
||||
ParseError: If frontmatter is invalid or parsing fails
|
||||
"""
|
||||
try:
|
||||
# Strip BOM before parsing frontmatter
|
||||
content = strip_bom(content)
|
||||
if not content.strip().startswith("---"):
|
||||
raise ParseError("Content has no frontmatter")
|
||||
|
||||
@@ -159,7 +367,8 @@ def remove_frontmatter(content: str) -> str:
|
||||
Raises:
|
||||
ParseError: If content starts with frontmatter marker but is malformed
|
||||
"""
|
||||
content = content.strip()
|
||||
# Strip BOM before processing
|
||||
content = strip_bom(content).strip()
|
||||
|
||||
# Return as-is if no frontmatter marker
|
||||
if not content.startswith("---"):
|
||||
|
||||
@@ -5,15 +5,18 @@ from datetime import datetime
|
||||
from typing import Any
|
||||
|
||||
|
||||
def clean_filename(name: str) -> str: # pragma: no cover
|
||||
def clean_filename(name: str | None) -> str: # pragma: no cover
|
||||
"""Clean a string to be used as a filename.
|
||||
|
||||
Args:
|
||||
name: The string to clean.
|
||||
name: The string to clean (can be None).
|
||||
|
||||
Returns:
|
||||
A cleaned string suitable for use as a filename.
|
||||
"""
|
||||
# Handle None or empty input
|
||||
if not name:
|
||||
return "untitled"
|
||||
# Replace common punctuation and whitespace with underscores
|
||||
name = re.sub(r"[\s\-,.:/\\\[\]\(\)]+", "_", name)
|
||||
# Remove any non-alphanumeric or underscore characters
|
||||
|
||||
@@ -23,6 +23,7 @@ from basic_memory.markdown.schemas import (
|
||||
)
|
||||
from basic_memory.utils import parse_tags
|
||||
|
||||
|
||||
md = MarkdownIt().use(observation_plugin).use(relation_plugin)
|
||||
|
||||
|
||||
@@ -226,6 +227,12 @@ class EntityParser:
|
||||
Returns:
|
||||
EntityMarkdown with parsed content
|
||||
"""
|
||||
# Strip BOM before parsing (can be present in files from Windows or certain sources)
|
||||
# See issue #452
|
||||
from basic_memory.file_utils import strip_bom
|
||||
|
||||
content = strip_bom(content)
|
||||
|
||||
# Parse frontmatter with proper error handling for malformed YAML
|
||||
try:
|
||||
post = frontmatter.loads(content)
|
||||
|
||||
@@ -1,15 +1,19 @@
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
from typing import TYPE_CHECKING, Optional
|
||||
from collections import OrderedDict
|
||||
|
||||
from frontmatter import Post
|
||||
from loguru import logger
|
||||
|
||||
|
||||
from basic_memory import file_utils
|
||||
from basic_memory.file_utils import dump_frontmatter
|
||||
from basic_memory.markdown.entity_parser import EntityParser
|
||||
from basic_memory.markdown.schemas import EntityMarkdown, Observation, Relation
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from basic_memory.config import BasicMemoryConfig
|
||||
|
||||
|
||||
class DirtyFileError(Exception):
|
||||
"""Raised when attempting to write to a file that has been modified."""
|
||||
@@ -35,9 +39,14 @@ class MarkdownProcessor:
|
||||
3. Track schema changes (that's done by the database)
|
||||
"""
|
||||
|
||||
def __init__(self, entity_parser: EntityParser):
|
||||
"""Initialize processor with base path and parser."""
|
||||
def __init__(
|
||||
self,
|
||||
entity_parser: EntityParser,
|
||||
app_config: Optional["BasicMemoryConfig"] = None,
|
||||
):
|
||||
"""Initialize processor with parser and optional config."""
|
||||
self.entity_parser = entity_parser
|
||||
self.app_config = app_config
|
||||
|
||||
async def read_file(self, path: Path) -> EntityMarkdown:
|
||||
"""Read and parse file into EntityMarkdown schema.
|
||||
@@ -122,7 +131,17 @@ class MarkdownProcessor:
|
||||
# Write atomically and return checksum of updated file
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
await file_utils.write_file_atomic(path, final_content)
|
||||
return await file_utils.compute_checksum(final_content)
|
||||
|
||||
# Format file if configured (MarkdownProcessor always handles markdown files)
|
||||
content_for_checksum = final_content
|
||||
if self.app_config:
|
||||
formatted_content = await file_utils.format_file(
|
||||
path, self.app_config, is_markdown=True
|
||||
)
|
||||
if formatted_content is not None:
|
||||
content_for_checksum = formatted_content
|
||||
|
||||
return await file_utils.compute_checksum(content_for_checksum)
|
||||
|
||||
def format_observations(self, observations: list[Observation]) -> str:
|
||||
"""Format observations section in standard way.
|
||||
|
||||
@@ -30,7 +30,9 @@ def is_observation(token: Token) -> bool:
|
||||
|
||||
# Check for proper observation format: [category] content
|
||||
match = re.match(r"^\[([^\[\]()]+)\]\s+(.+)", content)
|
||||
has_tags = "#" in content
|
||||
# Check for standalone hashtags (words starting with #)
|
||||
# This excludes # in HTML attributes like color="#4285F4"
|
||||
has_tags = any(part.startswith("#") for part in content.split())
|
||||
return bool(match) or has_tags
|
||||
|
||||
|
||||
@@ -160,7 +162,7 @@ def parse_inline_relations(content: str) -> List[Dict[str, Any]]:
|
||||
|
||||
target = content[start + 2 : end].strip()
|
||||
if target:
|
||||
relations.append({"type": "links to", "target": target, "context": None})
|
||||
relations.append({"type": "links_to", "target": target, "context": None})
|
||||
|
||||
start = end + 2
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
|
||||
from frontmatter import Post
|
||||
|
||||
from basic_memory.file_utils import has_frontmatter, remove_frontmatter, parse_frontmatter
|
||||
@@ -12,7 +13,10 @@ from basic_memory.models import Observation as ObservationModel
|
||||
|
||||
|
||||
def entity_model_from_markdown(
|
||||
file_path: Path, markdown: EntityMarkdown, entity: Optional[Entity] = None
|
||||
file_path: Path,
|
||||
markdown: EntityMarkdown,
|
||||
entity: Optional[Entity] = None,
|
||||
project_id: Optional[int] = None,
|
||||
) -> Entity:
|
||||
"""
|
||||
Convert markdown entity to model. Does not include relations.
|
||||
@@ -21,6 +25,7 @@ def entity_model_from_markdown(
|
||||
file_path: Path to the markdown file
|
||||
markdown: Parsed markdown entity
|
||||
entity: Optional existing entity to update
|
||||
project_id: Project ID for new observations (uses entity.project_id if not provided)
|
||||
|
||||
Returns:
|
||||
Entity model populated from markdown
|
||||
@@ -50,9 +55,13 @@ def entity_model_from_markdown(
|
||||
metadata = markdown.frontmatter.metadata or {}
|
||||
model.entity_metadata = {k: str(v) for k, v in metadata.items() if v is not None}
|
||||
|
||||
# Get project_id from entity if not provided
|
||||
obs_project_id = project_id or (model.project_id if hasattr(model, "project_id") else None)
|
||||
|
||||
# Convert observations
|
||||
model.observations = [
|
||||
ObservationModel(
|
||||
project_id=obs_project_id,
|
||||
content=obs.content,
|
||||
category=obs.category,
|
||||
context=obs.context,
|
||||
|
||||
@@ -95,6 +95,7 @@ async def get_client() -> AsyncIterator[AsyncClient]:
|
||||
yield client
|
||||
else:
|
||||
# Local mode: ASGI transport for in-process calls
|
||||
# Note: ASGI transport does NOT trigger FastAPI lifespan, so no special handling needed
|
||||
logger.info("Creating ASGI client for local Basic Memory API")
|
||||
async with AsyncClient(
|
||||
transport=ASGITransport(app=fastapi_app), base_url="http://test", timeout=timeout
|
||||
|
||||
@@ -2,8 +2,80 @@
|
||||
Basic Memory FastMCP server.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from fastmcp import FastMCP
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.services.initialization import initialize_app, initialize_file_sync
|
||||
from basic_memory.telemetry import show_notice_if_needed, track_app_started
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastMCP):
|
||||
"""Lifecycle manager for the MCP server.
|
||||
|
||||
Handles:
|
||||
- Database initialization and migrations
|
||||
- Telemetry notice and tracking
|
||||
- File sync in background (if enabled and not in cloud mode)
|
||||
- Proper cleanup on shutdown
|
||||
"""
|
||||
app_config = ConfigManager().config
|
||||
logger.info("Starting Basic Memory MCP server")
|
||||
|
||||
# Show telemetry notice (first run only) and track startup
|
||||
show_notice_if_needed()
|
||||
track_app_started("mcp")
|
||||
|
||||
# Track if we created the engine (vs test fixtures providing it)
|
||||
# This prevents disposing an engine provided by test fixtures when
|
||||
# multiple Client connections are made in the same test
|
||||
engine_was_none = db._engine is None
|
||||
|
||||
# Initialize app (runs migrations, reconciles projects)
|
||||
await initialize_app(app_config)
|
||||
|
||||
# Start file sync as background task (if enabled and not in cloud mode)
|
||||
sync_task = None
|
||||
if app_config.is_test_env:
|
||||
logger.info("Test environment detected - skipping local file sync")
|
||||
elif app_config.sync_changes and not app_config.cloud_mode_enabled:
|
||||
logger.info("Starting file sync in background")
|
||||
|
||||
async def _file_sync_runner() -> None:
|
||||
await initialize_file_sync(app_config)
|
||||
|
||||
sync_task = asyncio.create_task(_file_sync_runner())
|
||||
elif app_config.cloud_mode_enabled:
|
||||
logger.info("Cloud mode enabled - skipping local file sync")
|
||||
else:
|
||||
logger.info("Sync changes disabled - skipping file sync")
|
||||
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
# Shutdown
|
||||
logger.info("Shutting down Basic Memory MCP server")
|
||||
if sync_task:
|
||||
sync_task.cancel()
|
||||
try:
|
||||
await sync_task
|
||||
except asyncio.CancelledError:
|
||||
logger.info("File sync task cancelled")
|
||||
|
||||
# Only shutdown DB if we created it (not if test fixture provided it)
|
||||
if engine_was_none:
|
||||
await db.shutdown_db()
|
||||
logger.info("Database connections closed")
|
||||
else:
|
||||
logger.debug("Skipping DB shutdown - engine provided externally")
|
||||
|
||||
|
||||
mcp = FastMCP(
|
||||
name="Basic Memory",
|
||||
lifespan=lifespan,
|
||||
)
|
||||
|
||||
@@ -9,6 +9,7 @@ from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas.base import TimeFrame
|
||||
from basic_memory.schemas.memory import (
|
||||
GraphContext,
|
||||
@@ -87,6 +88,7 @@ async def build_context(
|
||||
Raises:
|
||||
ToolError: If project doesn't exist or depth parameter is invalid
|
||||
"""
|
||||
track_mcp_tool("build_context")
|
||||
logger.info(f"Building context from {url} in project {project}")
|
||||
|
||||
# Convert string depth to integer if needed
|
||||
@@ -104,11 +106,9 @@ async def build_context(
|
||||
# Get the active project using the new stateless approach
|
||||
active_project = await get_active_project(client, project, context)
|
||||
|
||||
project_url = active_project.project_url
|
||||
|
||||
response = await call_get(
|
||||
client,
|
||||
f"{project_url}/memory/{memory_url_path(url)}",
|
||||
f"/v2/projects/{active_project.id}/memory/{memory_url_path(url)}",
|
||||
params={
|
||||
"depth": depth,
|
||||
"timeframe": timeframe,
|
||||
|
||||
@@ -12,7 +12,8 @@ from fastmcp import Context
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_put
|
||||
from basic_memory.mcp.tools.utils import call_put, call_post, resolve_entity_id
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
@@ -94,9 +95,9 @@ async def canvas(
|
||||
Raises:
|
||||
ToolError: If project doesn't exist or folder path is invalid
|
||||
"""
|
||||
track_mcp_tool("canvas")
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
# Ensure path has .canvas extension
|
||||
file_title = title if title.endswith(".canvas") else f"{title}.canvas"
|
||||
@@ -108,23 +109,44 @@ async def canvas(
|
||||
# Convert to JSON
|
||||
canvas_json = json.dumps(canvas_data, indent=2)
|
||||
|
||||
# Write the file using the resource API
|
||||
# Try to create the canvas file first (optimistic create)
|
||||
logger.info(f"Creating canvas file: {file_path} in project {project}")
|
||||
# Send canvas_json as content string, not as json parameter
|
||||
# The resource endpoint expects Body() string content, not JSON-encoded data
|
||||
response = await call_put(
|
||||
client,
|
||||
f"{project_url}/resource/{file_path}",
|
||||
content=canvas_json,
|
||||
headers={"Content-Type": "text/plain"},
|
||||
)
|
||||
try:
|
||||
response = await call_post(
|
||||
client,
|
||||
f"/v2/projects/{active_project.id}/resource",
|
||||
json={"file_path": file_path, "content": canvas_json},
|
||||
)
|
||||
action = "Created"
|
||||
except Exception as e:
|
||||
# If creation failed due to conflict (already exists), try to update
|
||||
if (
|
||||
"409" in str(e)
|
||||
or "conflict" in str(e).lower()
|
||||
or "already exists" in str(e).lower()
|
||||
):
|
||||
logger.info(f"Canvas file exists, updating instead: {file_path}")
|
||||
try:
|
||||
entity_id = await resolve_entity_id(client, active_project.id, file_path)
|
||||
# For update, send content in JSON body
|
||||
response = await call_put(
|
||||
client,
|
||||
f"/v2/projects/{active_project.id}/resource/{entity_id}",
|
||||
json={"content": canvas_json},
|
||||
)
|
||||
action = "Updated"
|
||||
except Exception as update_error:
|
||||
# Re-raise the original error if update also fails
|
||||
raise e from update_error
|
||||
else:
|
||||
# Re-raise if it's not a conflict error
|
||||
raise
|
||||
|
||||
# Parse response
|
||||
result = response.json()
|
||||
logger.debug(result)
|
||||
|
||||
# Build summary
|
||||
action = "Created" if response.status_code == 201 else "Updated"
|
||||
summary = [f"# {action}: {file_path}", "\nThe canvas is ready to open in Obsidian."]
|
||||
|
||||
return "\n".join(summary)
|
||||
|
||||
@@ -15,6 +15,7 @@ from basic_memory.mcp.tools.search import search_notes
|
||||
from basic_memory.mcp.tools.read_note import read_note
|
||||
from basic_memory.schemas.search import SearchResponse
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
|
||||
|
||||
def _format_search_results_for_chatgpt(results: SearchResponse) -> List[Dict[str, Any]]:
|
||||
@@ -88,6 +89,7 @@ async def search(
|
||||
List with one dict: `{ "type": "text", "text": "{...JSON...}" }`
|
||||
where the JSON body contains `results`, `total_count`, and echo of `query`.
|
||||
"""
|
||||
track_mcp_tool("search")
|
||||
logger.info(f"ChatGPT search request: query='{query}'")
|
||||
|
||||
try:
|
||||
@@ -151,6 +153,7 @@ async def fetch(
|
||||
List with one dict: `{ "type": "text", "text": "{...JSON...}" }`
|
||||
where the JSON body includes `id`, `title`, `text`, `url`, and metadata.
|
||||
"""
|
||||
track_mcp_tool("fetch")
|
||||
logger.info(f"ChatGPT fetch request: id='{id}'")
|
||||
|
||||
try:
|
||||
|
||||
@@ -3,11 +3,13 @@ from typing import Optional
|
||||
|
||||
from loguru import logger
|
||||
from fastmcp import Context
|
||||
from mcp.server.fastmcp.exceptions import ToolError
|
||||
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.tools.utils import call_delete
|
||||
from basic_memory.mcp.tools.utils import call_delete, resolve_entity_id
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas import DeleteEntitiesResponse
|
||||
|
||||
|
||||
@@ -202,12 +204,27 @@ async def delete_note(
|
||||
with suggestions for finding the correct identifier, including search
|
||||
commands and alternative formats to try.
|
||||
"""
|
||||
track_mcp_tool("delete_note")
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
try:
|
||||
response = await call_delete(client, f"{project_url}/knowledge/entities/{identifier}")
|
||||
# Resolve identifier to entity ID
|
||||
entity_id = await resolve_entity_id(client, active_project.id, identifier)
|
||||
except ToolError as e:
|
||||
# If entity not found, return False (note doesn't exist)
|
||||
if "Entity not found" in str(e) or "not found" in str(e).lower():
|
||||
logger.warning(f"Note not found for deletion: {identifier}")
|
||||
return False
|
||||
# For other resolution errors, return formatted error message
|
||||
logger.error(f"Delete failed for '{identifier}': {e}, project: {active_project.name}")
|
||||
return _format_delete_error_response(active_project.name, str(e), identifier)
|
||||
|
||||
try:
|
||||
# Call the DELETE endpoint
|
||||
response = await call_delete(
|
||||
client, f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}"
|
||||
)
|
||||
result = DeleteEntitiesResponse.model_validate(response.json())
|
||||
|
||||
if result.deleted:
|
||||
|
||||
@@ -8,7 +8,8 @@ from fastmcp import Context
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project, add_project_metadata
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_patch
|
||||
from basic_memory.mcp.tools.utils import call_patch, resolve_entity_id
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas import EntityResponse
|
||||
|
||||
|
||||
@@ -214,9 +215,9 @@ async def edit_note(
|
||||
search_notes() first to find the correct identifier. The tool provides detailed
|
||||
error messages with suggestions if operations fail.
|
||||
"""
|
||||
track_mcp_tool("edit_note")
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
logger.info("MCP tool call", tool="edit_note", identifier=identifier, operation=operation)
|
||||
|
||||
@@ -235,6 +236,9 @@ async def edit_note(
|
||||
|
||||
# Use the PATCH endpoint to edit the entity
|
||||
try:
|
||||
# Resolve identifier to entity ID
|
||||
entity_id = await resolve_entity_id(client, active_project.id, identifier)
|
||||
|
||||
# Prepare the edit request data
|
||||
edit_data = {
|
||||
"operation": operation,
|
||||
@@ -250,7 +254,7 @@ async def edit_note(
|
||||
edit_data["expected_replacements"] = str(expected_replacements)
|
||||
|
||||
# Call the PATCH endpoint
|
||||
url = f"{project_url}/knowledge/entities/{identifier}"
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}"
|
||||
response = await call_patch(client, url, json=edit_data)
|
||||
result = EntityResponse.model_validate(response.json())
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
@@ -63,9 +64,9 @@ async def list_directory(
|
||||
Raises:
|
||||
ToolError: If project doesn't exist or directory path is invalid
|
||||
"""
|
||||
track_mcp_tool("list_directory")
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
# Prepare query parameters
|
||||
params = {
|
||||
@@ -82,7 +83,7 @@ async def list_directory(
|
||||
# Call the API endpoint
|
||||
response = await call_get(
|
||||
client,
|
||||
f"{project_url}/directory/list",
|
||||
f"/v2/projects/{active_project.id}/directory/list",
|
||||
params=params,
|
||||
)
|
||||
|
||||
|
||||
@@ -8,10 +8,11 @@ from fastmcp import Context
|
||||
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_post, call_get
|
||||
from basic_memory.mcp.tools.utils import call_get, call_put, resolve_entity_id
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.schemas import EntityResponse
|
||||
from basic_memory.schemas.project_info import ProjectList
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.utils import validate_project_path
|
||||
|
||||
|
||||
@@ -395,11 +396,11 @@ async def move_note(
|
||||
- Re-indexes the entity for search
|
||||
- Maintains all observations and relations
|
||||
"""
|
||||
track_mcp_tool("move_note")
|
||||
async with get_client() as client:
|
||||
logger.debug(f"Moving note: {identifier} to {destination_path} in project: {project}")
|
||||
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
# Validate destination path to prevent path traversal attacks
|
||||
project_path = active_project.home
|
||||
@@ -434,8 +435,10 @@ move_note("{identifier}", "notes/{destination_path.split("/")[-1] if "/" in dest
|
||||
# Get the source entity information for extension validation
|
||||
source_ext = "md" # Default to .md if we can't determine source extension
|
||||
try:
|
||||
# Resolve identifier to entity ID
|
||||
entity_id = await resolve_entity_id(client, active_project.id, identifier)
|
||||
# Fetch source entity information to get the current file extension
|
||||
url = f"{project_url}/knowledge/entities/{identifier}"
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}"
|
||||
response = await call_get(client, url)
|
||||
source_entity = EntityResponse.model_validate(response.json())
|
||||
if "." in source_entity.file_path:
|
||||
@@ -467,8 +470,10 @@ move_note("{identifier}", "notes/{destination_path.split("/")[-1] if "/" in dest
|
||||
|
||||
# Get the source entity to check its file extension
|
||||
try:
|
||||
# Resolve identifier to entity ID (might already be cached from above)
|
||||
entity_id = await resolve_entity_id(client, active_project.id, identifier)
|
||||
# Fetch source entity information
|
||||
url = f"{project_url}/knowledge/entities/{identifier}"
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}"
|
||||
response = await call_get(client, url)
|
||||
source_entity = EntityResponse.model_validate(response.json())
|
||||
|
||||
@@ -505,16 +510,17 @@ move_note("{identifier}", "notes/{destination_path.split("/")[-1] if "/" in dest
|
||||
logger.debug(f"Could not fetch source entity for extension check: {e}")
|
||||
|
||||
try:
|
||||
# Prepare move request
|
||||
# Resolve identifier to entity ID for the move operation
|
||||
entity_id = await resolve_entity_id(client, active_project.id, identifier)
|
||||
|
||||
# Prepare move request (v2 API only needs destination_path)
|
||||
move_data = {
|
||||
"identifier": identifier,
|
||||
"destination_path": destination_path,
|
||||
"project": active_project.name,
|
||||
}
|
||||
|
||||
# Call the move API endpoint
|
||||
url = f"{project_url}/knowledge/move"
|
||||
response = await call_post(client, url, json=move_data)
|
||||
# Call the v2 move API endpoint (PUT method, entity_id in URL)
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}/move"
|
||||
response = await call_put(client, url, json=move_data)
|
||||
result = EntityResponse.model_validate(response.json())
|
||||
|
||||
# Build success message
|
||||
|
||||
@@ -15,6 +15,7 @@ from basic_memory.schemas.project_info import (
|
||||
ProjectStatusResponse,
|
||||
ProjectInfoRequest,
|
||||
)
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
|
||||
@@ -40,6 +41,7 @@ async def list_memory_projects(context: Context | None = None) -> str:
|
||||
Example:
|
||||
list_memory_projects()
|
||||
"""
|
||||
track_mcp_tool("list_memory_projects")
|
||||
async with get_client() as client:
|
||||
if context: # pragma: no cover
|
||||
await context.info("Listing all available projects")
|
||||
@@ -92,6 +94,7 @@ async def create_memory_project(
|
||||
create_memory_project("my-research", "~/Documents/research")
|
||||
create_memory_project("work-notes", "/home/user/work", set_default=True)
|
||||
"""
|
||||
track_mcp_tool("create_memory_project")
|
||||
async with get_client() as client:
|
||||
# Check if server is constrained to a specific project
|
||||
constrained_project = os.environ.get("BASIC_MEMORY_MCP_PROJECT")
|
||||
@@ -147,6 +150,7 @@ async def delete_project(project_name: str, context: Context | None = None) -> s
|
||||
This action cannot be undone. The project will need to be re-added
|
||||
to access its content through Basic Memory again.
|
||||
"""
|
||||
track_mcp_tool("delete_project")
|
||||
async with get_client() as client:
|
||||
# Check if server is constrained to a specific project
|
||||
constrained_project = os.environ.get("BASIC_MEMORY_MCP_PROJECT")
|
||||
@@ -179,11 +183,8 @@ async def delete_project(project_name: str, context: Context | None = None) -> s
|
||||
f"Project '{project_name}' not found. Available projects: {', '.join(available_projects)}"
|
||||
)
|
||||
|
||||
# Call API to delete project using URL encoding for special characters
|
||||
from urllib.parse import quote
|
||||
|
||||
encoded_name = quote(target_project.name, safe="")
|
||||
response = await call_delete(client, f"/projects/{encoded_name}")
|
||||
# Call v2 API to delete project using project ID
|
||||
response = await call_delete(client, f"/v2/projects/{target_project.id}")
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
|
||||
result = f"✓ {status_response.message}\n\n"
|
||||
|
||||
@@ -13,12 +13,14 @@ from typing import Optional
|
||||
from loguru import logger
|
||||
from PIL import Image as PILImage
|
||||
from fastmcp import Context
|
||||
from mcp.server.fastmcp.exceptions import ToolError
|
||||
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.mcp.tools.utils import call_get, resolve_entity_id
|
||||
from basic_memory.schemas.memory import memory_url_path
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.utils import validate_project_path
|
||||
|
||||
|
||||
@@ -199,11 +201,11 @@ async def read_content(
|
||||
HTTPError: If project doesn't exist or is inaccessible
|
||||
SecurityError: If path attempts path traversal
|
||||
"""
|
||||
track_mcp_tool("read_content")
|
||||
logger.info("Reading file", path=path, project=project)
|
||||
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
url = memory_url_path(path)
|
||||
|
||||
@@ -221,7 +223,15 @@ async def read_content(
|
||||
"error": f"Path '{path}' is not allowed - paths must stay within project boundaries",
|
||||
}
|
||||
|
||||
response = await call_get(client, f"{project_url}/resource/{url}")
|
||||
# Resolve path to entity ID
|
||||
try:
|
||||
entity_id = await resolve_entity_id(client, active_project.id, url)
|
||||
except ToolError:
|
||||
# Convert resolution errors to "Resource not found" for consistency
|
||||
raise ToolError(f"Resource not found: {url}")
|
||||
|
||||
# Call the v2 resource endpoint
|
||||
response = await call_get(client, f"/v2/projects/{active_project.id}/resource/{entity_id}")
|
||||
content_type = response.headers.get("content-type", "application/octet-stream")
|
||||
content_length = int(response.headers.get("content-length", 0))
|
||||
|
||||
|
||||
@@ -10,7 +10,8 @@ from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.search import search_notes
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.mcp.tools.utils import call_get, resolve_entity_id
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas.memory import memory_url_path
|
||||
from basic_memory.utils import validate_project_path
|
||||
|
||||
@@ -77,6 +78,7 @@ async def read_note(
|
||||
If the exact note isn't found, this tool provides helpful suggestions
|
||||
including related notes, search commands, and note creation templates.
|
||||
"""
|
||||
track_mcp_tool("read_note")
|
||||
async with get_client() as client:
|
||||
# Get and validate the project
|
||||
active_project = await get_active_project(client, project, context)
|
||||
@@ -97,23 +99,29 @@ async def read_note(
|
||||
)
|
||||
return f"# Error\n\nIdentifier '{identifier}' is not allowed - paths must stay within project boundaries"
|
||||
|
||||
project_url = active_project.project_url
|
||||
|
||||
# Get the file via REST API - first try direct permalink lookup
|
||||
# Get the file via REST API - first try direct identifier resolution
|
||||
entity_path = memory_url_path(identifier)
|
||||
path = f"{project_url}/resource/{entity_path}"
|
||||
logger.info(f"Attempting to read note from Project: {active_project.name} URL: {path}")
|
||||
logger.info(
|
||||
f"Attempting to read note from Project: {active_project.name} identifier: {entity_path}"
|
||||
)
|
||||
|
||||
try:
|
||||
# Try direct lookup first
|
||||
response = await call_get(client, path, params={"page": page, "page_size": page_size})
|
||||
# Try to resolve identifier to entity ID
|
||||
entity_id = await resolve_entity_id(client, active_project.id, entity_path)
|
||||
|
||||
# Fetch content using entity ID
|
||||
response = await call_get(
|
||||
client,
|
||||
f"/v2/projects/{active_project.id}/resource/{entity_id}",
|
||||
params={"page": page, "page_size": page_size},
|
||||
)
|
||||
|
||||
# If successful, return the content
|
||||
if response.status_code == 200:
|
||||
logger.info("Returning read_note result from resource: {path}", path=entity_path)
|
||||
return response.text
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.info(f"Direct lookup failed for '{path}': {e}")
|
||||
logger.info(f"Direct lookup failed for '{entity_path}': {e}")
|
||||
# Continue to fallback methods
|
||||
|
||||
# Fallback 1: Try title search via API
|
||||
@@ -127,10 +135,14 @@ async def read_note(
|
||||
result = title_results.results[0] # Get the first/best match
|
||||
if result.permalink:
|
||||
try:
|
||||
# Try to fetch the content using the found permalink
|
||||
path = f"{project_url}/resource/{result.permalink}"
|
||||
# Resolve the permalink to entity ID
|
||||
entity_id = await resolve_entity_id(client, active_project.id, result.permalink)
|
||||
|
||||
# Fetch content using the entity ID
|
||||
response = await call_get(
|
||||
client, path, params={"page": page, "page_size": page_size}
|
||||
client,
|
||||
f"/v2/projects/{active_project.id}/resource/{entity_id}",
|
||||
params={"page": page, "page_size": page_size},
|
||||
)
|
||||
|
||||
if response.status_code == 200:
|
||||
|
||||
@@ -9,6 +9,7 @@ from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project, resolve_project_parameter
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_get
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas.base import TimeFrame
|
||||
from basic_memory.schemas.memory import (
|
||||
GraphContext,
|
||||
@@ -98,6 +99,7 @@ async def recent_activity(
|
||||
- For focused queries, consider using build_context with a specific URI
|
||||
- Max timeframe is 1 year in the past
|
||||
"""
|
||||
track_mcp_tool("recent_activity")
|
||||
async with get_client() as client:
|
||||
# Build common parameters for API calls
|
||||
params = {
|
||||
@@ -247,11 +249,10 @@ async def recent_activity(
|
||||
)
|
||||
|
||||
active_project = await get_active_project(client, resolved_project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
response = await call_get(
|
||||
client,
|
||||
f"{project_url}/memory/recent",
|
||||
f"/v2/projects/{active_project.id}/memory/recent",
|
||||
params=params,
|
||||
)
|
||||
activity_data = GraphContext.model_validate(response.json())
|
||||
@@ -274,10 +275,9 @@ async def _get_project_activity(
|
||||
Returns:
|
||||
ProjectActivity with activity data or empty activity on error
|
||||
"""
|
||||
project_url = f"/{project_info.permalink}"
|
||||
activity_response = await call_get(
|
||||
client,
|
||||
f"{project_url}/memory/recent",
|
||||
f"/v2/projects/{project_info.id}/memory/recent",
|
||||
params=params,
|
||||
)
|
||||
activity = GraphContext.model_validate(activity_response.json())
|
||||
|
||||
@@ -10,6 +10,7 @@ from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_post
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas.search import SearchItemType, SearchQuery, SearchResponse
|
||||
|
||||
|
||||
@@ -330,6 +331,7 @@ async def search_notes(
|
||||
# Explicit project specification
|
||||
results = await search_notes("project planning", project="my-project")
|
||||
"""
|
||||
track_mcp_tool("search_notes")
|
||||
# Create a SearchQuery object based on the parameters
|
||||
search_query = SearchQuery()
|
||||
|
||||
@@ -355,14 +357,13 @@ async def search_notes(
|
||||
|
||||
async with get_client() as client:
|
||||
active_project = await get_active_project(client, project, context)
|
||||
project_url = active_project.project_url
|
||||
|
||||
logger.info(f"Searching for {search_query} in project {active_project.name}")
|
||||
|
||||
try:
|
||||
response = await call_post(
|
||||
client,
|
||||
f"{project_url}/search/",
|
||||
f"/v2/projects/{active_project.id}/search/",
|
||||
json=search_query.model_dump(),
|
||||
params={"page": page, "page_size": page_size},
|
||||
)
|
||||
|
||||
@@ -435,6 +435,34 @@ async def call_post(
|
||||
raise ToolError(error_message) from e
|
||||
|
||||
|
||||
async def resolve_entity_id(client: AsyncClient, project_id: int, identifier: str) -> int:
|
||||
"""Resolve a string identifier to an entity ID using the v2 API.
|
||||
|
||||
Args:
|
||||
client: HTTP client for API calls
|
||||
project_id: Project ID
|
||||
identifier: The identifier to resolve (permalink, title, or path)
|
||||
|
||||
Returns:
|
||||
The resolved entity ID
|
||||
|
||||
Raises:
|
||||
ToolError: If the identifier cannot be resolved
|
||||
"""
|
||||
try:
|
||||
response = await call_post(
|
||||
client, f"/v2/projects/{project_id}/knowledge/resolve", json={"identifier": identifier}
|
||||
)
|
||||
data = response.json()
|
||||
return data["entity_id"]
|
||||
except HTTPStatusError as e:
|
||||
if e.response.status_code == 404:
|
||||
raise ToolError(f"Entity not found: '{identifier}'")
|
||||
raise ToolError(f"Error resolving identifier '{identifier}': {e}")
|
||||
except Exception as e:
|
||||
raise ToolError(f"Unexpected error resolving identifier '{identifier}': {e}")
|
||||
|
||||
|
||||
async def call_delete(
|
||||
client: AsyncClient,
|
||||
url: URL | str,
|
||||
|
||||
@@ -8,6 +8,7 @@ from fastmcp import Context
|
||||
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.read_note import read_note
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
@@ -54,7 +55,7 @@ async def view_note(
|
||||
HTTPError: If project doesn't exist or is inaccessible
|
||||
SecurityError: If identifier attempts path traversal
|
||||
"""
|
||||
|
||||
track_mcp_tool("view_note")
|
||||
logger.info(f"Viewing note: {identifier} in project: {project}")
|
||||
|
||||
# Call the existing read_note logic
|
||||
|
||||
@@ -7,7 +7,8 @@ from loguru import logger
|
||||
from basic_memory.mcp.async_client import get_client
|
||||
from basic_memory.mcp.project_context import get_active_project, add_project_metadata
|
||||
from basic_memory.mcp.server import mcp
|
||||
from basic_memory.mcp.tools.utils import call_put
|
||||
from basic_memory.mcp.tools.utils import call_put, call_post, resolve_entity_id
|
||||
from basic_memory.telemetry import track_mcp_tool
|
||||
from basic_memory.schemas import EntityResponse
|
||||
from fastmcp import Context
|
||||
from basic_memory.schemas.base import Entity
|
||||
@@ -116,6 +117,7 @@ async def write_note(
|
||||
HTTPError: If project doesn't exist or is inaccessible
|
||||
SecurityError: If folder path attempts path traversal
|
||||
"""
|
||||
track_mcp_tool("write_note")
|
||||
async with get_client() as client:
|
||||
logger.info(
|
||||
f"MCP tool call tool=write_note project={project} folder={folder}, title={title}, tags={tags}"
|
||||
@@ -150,16 +152,37 @@ async def write_note(
|
||||
content=content,
|
||||
entity_metadata=metadata,
|
||||
)
|
||||
project_url = active_project.permalink
|
||||
|
||||
# Create or update via knowledge API
|
||||
logger.debug(f"Creating entity via API permalink={entity.permalink}")
|
||||
url = f"{project_url}/knowledge/entities/{entity.permalink}"
|
||||
response = await call_put(client, url, json=entity.model_dump())
|
||||
result = EntityResponse.model_validate(response.json())
|
||||
|
||||
# Format semantic summary based on status code
|
||||
action = "Created" if response.status_code == 201 else "Updated"
|
||||
# Try to create the entity first (optimistic create)
|
||||
logger.debug(f"Attempting to create entity permalink={entity.permalink}")
|
||||
action = "Created" # Default to created
|
||||
try:
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities"
|
||||
response = await call_post(client, url, json=entity.model_dump())
|
||||
result = EntityResponse.model_validate(response.json())
|
||||
action = "Created"
|
||||
except Exception as e:
|
||||
# If creation failed due to conflict (already exists), try to update
|
||||
if (
|
||||
"409" in str(e)
|
||||
or "conflict" in str(e).lower()
|
||||
or "already exists" in str(e).lower()
|
||||
):
|
||||
logger.debug(f"Entity exists, updating instead permalink={entity.permalink}")
|
||||
try:
|
||||
if not entity.permalink:
|
||||
raise ValueError("Entity permalink is required for updates")
|
||||
entity_id = await resolve_entity_id(client, active_project.id, entity.permalink)
|
||||
url = f"/v2/projects/{active_project.id}/knowledge/entities/{entity_id}"
|
||||
response = await call_put(client, url, json=entity.model_dump())
|
||||
result = EntityResponse.model_validate(response.json())
|
||||
action = "Updated"
|
||||
except Exception as update_error:
|
||||
# Re-raise the original error if update also fails
|
||||
raise e from update_error
|
||||
else:
|
||||
# Re-raise if it's not a conflict error
|
||||
raise
|
||||
summary = [
|
||||
f"# {action} note",
|
||||
f"project: {active_project.name}",
|
||||
|
||||
@@ -4,7 +4,6 @@ import basic_memory
|
||||
from basic_memory.models.base import Base
|
||||
from basic_memory.models.knowledge import Entity, Observation, Relation
|
||||
from basic_memory.models.project import Project
|
||||
from basic_memory.models.search import SearchIndex
|
||||
|
||||
__all__ = [
|
||||
"Base",
|
||||
@@ -12,6 +11,5 @@ __all__ = [
|
||||
"Observation",
|
||||
"Relation",
|
||||
"Project",
|
||||
"SearchIndex",
|
||||
"basic_memory",
|
||||
]
|
||||
|
||||
@@ -145,6 +145,7 @@ class Observation(Base):
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
project_id: Mapped[int] = mapped_column(Integer, ForeignKey("project.id"), index=True)
|
||||
entity_id: Mapped[int] = mapped_column(Integer, ForeignKey("entity.id", ondelete="CASCADE"))
|
||||
content: Mapped[str] = mapped_column(Text)
|
||||
category: Mapped[str] = mapped_column(String, nullable=False, default="note")
|
||||
@@ -162,9 +163,14 @@ class Observation(Base):
|
||||
|
||||
We can construct these because observations are always defined in
|
||||
and owned by a single entity.
|
||||
|
||||
Content is truncated to 200 chars to stay under PostgreSQL's
|
||||
btree index limit of 2704 bytes.
|
||||
"""
|
||||
# Truncate content to avoid exceeding PostgreSQL's btree index limit
|
||||
content_for_permalink = self.content[:200] if len(self.content) > 200 else self.content
|
||||
return generate_permalink(
|
||||
f"{self.entity.permalink}/observations/{self.category}/{self.content}"
|
||||
f"{self.entity.permalink}/observations/{self.category}/{content_for_permalink}"
|
||||
)
|
||||
|
||||
def __repr__(self) -> str: # pragma: no cover
|
||||
@@ -186,6 +192,7 @@ class Relation(Base):
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
project_id: Mapped[int] = mapped_column(Integer, ForeignKey("project.id"), index=True)
|
||||
from_id: Mapped[int] = mapped_column(Integer, ForeignKey("entity.id", ondelete="CASCADE"))
|
||||
to_id: Mapped[Optional[int]] = mapped_column(
|
||||
Integer, ForeignKey("entity.id", ondelete="CASCADE"), nullable=True
|
||||
|
||||
@@ -1,53 +1,52 @@
|
||||
"""Search models and tables."""
|
||||
"""Search DDL statements for SQLite and Postgres.
|
||||
|
||||
from sqlalchemy import DDL, Column, Integer, String, DateTime, Text
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
from sqlalchemy.types import JSON
|
||||
The search_index table is created via raw DDL, not ORM models, because:
|
||||
- SQLite uses FTS5 virtual tables (cannot be represented as ORM)
|
||||
- Postgres uses composite primary keys and generated tsvector columns
|
||||
- Both backends use raw SQL for all search operations via SearchIndexRow dataclass
|
||||
"""
|
||||
|
||||
from basic_memory.models.base import Base
|
||||
from sqlalchemy import DDL
|
||||
|
||||
|
||||
class SearchIndex(Base):
|
||||
"""Search index table for Postgres only.
|
||||
# Define Postgres search_index table with composite primary key and tsvector
|
||||
# This DDL matches the Alembic migration schema (314f1ea54dc4)
|
||||
# Used by tests to create the table without running full migrations
|
||||
# NOTE: Split into separate DDL statements because asyncpg doesn't support
|
||||
# multiple statements in a single execute call.
|
||||
CREATE_POSTGRES_SEARCH_INDEX_TABLE = DDL("""
|
||||
CREATE TABLE IF NOT EXISTS search_index (
|
||||
id INTEGER NOT NULL,
|
||||
project_id INTEGER NOT NULL,
|
||||
title TEXT,
|
||||
content_stems TEXT,
|
||||
content_snippet TEXT,
|
||||
permalink VARCHAR,
|
||||
file_path VARCHAR,
|
||||
type VARCHAR,
|
||||
from_id INTEGER,
|
||||
to_id INTEGER,
|
||||
relation_type VARCHAR,
|
||||
entity_id INTEGER,
|
||||
category VARCHAR,
|
||||
metadata JSONB,
|
||||
created_at TIMESTAMP WITH TIME ZONE,
|
||||
updated_at TIMESTAMP WITH TIME ZONE,
|
||||
textsearchable_index_col tsvector GENERATED ALWAYS AS (
|
||||
to_tsvector('english', coalesce(title, '') || ' ' || coalesce(content_stems, ''))
|
||||
) STORED,
|
||||
PRIMARY KEY (id, type, project_id),
|
||||
FOREIGN KEY (project_id) REFERENCES project(id) ON DELETE CASCADE
|
||||
)
|
||||
""")
|
||||
|
||||
For SQLite: This model is skipped; FTS5 virtual table is created via DDL instead.
|
||||
For Postgres: This is the actual table structure with tsvector support.
|
||||
"""
|
||||
|
||||
__tablename__ = "search_index"
|
||||
|
||||
# Primary key (rowid in SQLite FTS5, explicit id in Postgres)
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
|
||||
# Core searchable fields
|
||||
title = Column(Text, nullable=True)
|
||||
content_stems = Column(Text, nullable=True)
|
||||
content_snippet = Column(Text, nullable=True)
|
||||
permalink = Column(String(255), nullable=True, index=True)
|
||||
file_path = Column(Text, nullable=True)
|
||||
type = Column(String(50), nullable=True)
|
||||
|
||||
# Project context
|
||||
project_id = Column(Integer, nullable=True, index=True)
|
||||
|
||||
# Relation fields
|
||||
from_id = Column(Integer, nullable=True)
|
||||
to_id = Column(Integer, nullable=True)
|
||||
relation_type = Column(String(100), nullable=True)
|
||||
|
||||
# Observation fields
|
||||
entity_id = Column(Integer, nullable=True)
|
||||
category = Column(String(100), nullable=True)
|
||||
|
||||
# Common fields
|
||||
# Use JSONB for Postgres, JSON for SQLite
|
||||
# Note: 'metadata' is a reserved name in SQLAlchemy, so we use 'metadata_' and map to 'metadata'
|
||||
metadata_ = Column("metadata", JSON().with_variant(JSONB(), "postgresql"), nullable=True)
|
||||
created_at = Column(DateTime(timezone=True), nullable=True)
|
||||
updated_at = Column(DateTime(timezone=True), nullable=True)
|
||||
|
||||
# Note: textsearchable_index_col (tsvector) will be added by migration for Postgres only
|
||||
CREATE_POSTGRES_SEARCH_INDEX_FTS = DDL("""
|
||||
CREATE INDEX IF NOT EXISTS idx_search_index_fts ON search_index USING gin(textsearchable_index_col)
|
||||
""")
|
||||
|
||||
CREATE_POSTGRES_SEARCH_INDEX_METADATA = DDL("""
|
||||
CREATE INDEX IF NOT EXISTS idx_search_index_metadata_gin ON search_index USING gin(metadata jsonb_path_ops)
|
||||
""")
|
||||
|
||||
# Define FTS5 virtual table creation for SQLite only
|
||||
# This DDL is executed separately for SQLite databases
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Sequence, Union, Any
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
@@ -76,6 +77,99 @@ class EntityRepository(Repository[Entity]):
|
||||
)
|
||||
return await self.find_one(query)
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Lightweight methods for permalink resolution (no eager loading)
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
async def permalink_exists(self, permalink: str) -> bool:
|
||||
"""Check if a permalink exists without loading the full entity.
|
||||
|
||||
This is much faster than get_by_permalink() as it skips eager loading
|
||||
of observations and relations. Use for existence checks in bulk operations.
|
||||
|
||||
Args:
|
||||
permalink: Permalink to check
|
||||
|
||||
Returns:
|
||||
True if permalink exists, False otherwise
|
||||
"""
|
||||
query = select(Entity.id).where(Entity.permalink == permalink).limit(1)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none() is not None
|
||||
|
||||
async def get_file_path_for_permalink(self, permalink: str) -> Optional[str]:
|
||||
"""Get the file_path for a permalink without loading the full entity.
|
||||
|
||||
Use when you only need the file_path, not the full entity with relations.
|
||||
|
||||
Args:
|
||||
permalink: Permalink to look up
|
||||
|
||||
Returns:
|
||||
file_path string if found, None otherwise
|
||||
"""
|
||||
query = select(Entity.file_path).where(Entity.permalink == permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none()
|
||||
|
||||
async def get_permalink_for_file_path(self, file_path: Union[Path, str]) -> Optional[str]:
|
||||
"""Get the permalink for a file_path without loading the full entity.
|
||||
|
||||
Use when you only need the permalink, not the full entity with relations.
|
||||
|
||||
Args:
|
||||
file_path: File path to look up
|
||||
|
||||
Returns:
|
||||
permalink string if found, None otherwise
|
||||
"""
|
||||
query = select(Entity.permalink).where(Entity.file_path == Path(file_path).as_posix())
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none()
|
||||
|
||||
async def get_all_permalinks(self) -> List[str]:
|
||||
"""Get all permalinks for this project.
|
||||
|
||||
Optimized for bulk operations - returns only permalink strings
|
||||
without loading entities or relationships.
|
||||
|
||||
Returns:
|
||||
List of all permalinks in the project
|
||||
"""
|
||||
query = select(Entity.permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return list(result.scalars().all())
|
||||
|
||||
async def get_permalink_to_file_path_map(self) -> dict[str, str]:
|
||||
"""Get a mapping of permalink -> file_path for all entities.
|
||||
|
||||
Optimized for bulk permalink resolution - loads minimal data in one query.
|
||||
|
||||
Returns:
|
||||
Dict mapping permalink to file_path
|
||||
"""
|
||||
query = select(Entity.permalink, Entity.file_path)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return {row.permalink: row.file_path for row in result.all()}
|
||||
|
||||
async def get_file_path_to_permalink_map(self) -> dict[str, str]:
|
||||
"""Get a mapping of file_path -> permalink for all entities.
|
||||
|
||||
Optimized for bulk permalink resolution - loads minimal data in one query.
|
||||
|
||||
Returns:
|
||||
Dict mapping file_path to permalink
|
||||
"""
|
||||
query = select(Entity.file_path, Entity.permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return {row.file_path: row.permalink for row in result.all()}
|
||||
|
||||
async def get_by_file_paths(
|
||||
self, session: AsyncSession, file_paths: Sequence[Union[Path, str]]
|
||||
) -> List[Row[Any]]:
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Dict, List, Sequence
|
||||
|
||||
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy.ext.asyncio import async_sessionmaker
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ import re
|
||||
from datetime import datetime
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
@@ -257,7 +258,7 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
{score_expr} as score
|
||||
FROM search_index
|
||||
WHERE {where_clause}
|
||||
ORDER BY score DESC {order_by_clause}
|
||||
ORDER BY score DESC, id ASC {order_by_clause}
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from pathlib import Path
|
||||
from typing import Optional, Sequence, Union
|
||||
|
||||
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
|
||||
@@ -23,7 +24,7 @@ class ProjectRepository(Repository[Project]):
|
||||
super().__init__(session_maker, Project)
|
||||
|
||||
async def get_by_name(self, name: str) -> Optional[Project]:
|
||||
"""Get project by name.
|
||||
"""Get project by name (exact match).
|
||||
|
||||
Args:
|
||||
name: Unique name of the project
|
||||
@@ -31,6 +32,18 @@ class ProjectRepository(Repository[Project]):
|
||||
query = self.select().where(Project.name == name)
|
||||
return await self.find_one(query)
|
||||
|
||||
async def get_by_name_case_insensitive(self, name: str) -> Optional[Project]:
|
||||
"""Get project by name (case-insensitive match).
|
||||
|
||||
Args:
|
||||
name: Project name (case-insensitive)
|
||||
|
||||
Returns:
|
||||
Project if found, None otherwise
|
||||
"""
|
||||
query = self.select().where(Project.name.ilike(name))
|
||||
return await self.find_one(query)
|
||||
|
||||
async def get_by_permalink(self, permalink: str) -> Optional[Project]:
|
||||
"""Get project by permalink.
|
||||
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
"""Repository for managing Relation objects."""
|
||||
|
||||
from sqlalchemy import and_, delete
|
||||
from typing import Sequence, List, Optional
|
||||
|
||||
from sqlalchemy import select
|
||||
|
||||
from sqlalchemy import and_, delete, select
|
||||
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||
from sqlalchemy.ext.asyncio import async_sessionmaker
|
||||
from sqlalchemy.orm import selectinload, aliased
|
||||
from sqlalchemy.orm.interfaces import LoaderOption
|
||||
@@ -86,5 +88,59 @@ class RelationRepository(Repository[Relation]):
|
||||
result = await self.execute_query(query)
|
||||
return result.scalars().all()
|
||||
|
||||
async def add_all_ignore_duplicates(self, relations: List[Relation]) -> int:
|
||||
"""Bulk insert relations, ignoring duplicates.
|
||||
|
||||
Uses ON CONFLICT DO NOTHING to skip relations that would violate the
|
||||
unique constraint on (from_id, to_name, relation_type). This is useful
|
||||
for bulk operations where the same link may appear multiple times in
|
||||
a document.
|
||||
|
||||
Works with both SQLite and PostgreSQL dialects.
|
||||
|
||||
Args:
|
||||
relations: List of Relation objects to insert
|
||||
|
||||
Returns:
|
||||
Number of relations actually inserted (excludes duplicates)
|
||||
"""
|
||||
if not relations:
|
||||
return 0
|
||||
|
||||
# Convert Relation objects to dicts for insert
|
||||
values = [
|
||||
{
|
||||
"project_id": r.project_id if r.project_id else self.project_id,
|
||||
"from_id": r.from_id,
|
||||
"to_id": r.to_id,
|
||||
"to_name": r.to_name,
|
||||
"relation_type": r.relation_type,
|
||||
"context": r.context,
|
||||
}
|
||||
for r in relations
|
||||
]
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Check dialect to use appropriate insert
|
||||
dialect_name = session.bind.dialect.name if session.bind else "sqlite"
|
||||
|
||||
if dialect_name == "postgresql":
|
||||
# PostgreSQL: use RETURNING to count inserted rows
|
||||
# (rowcount is 0 for ON CONFLICT DO NOTHING)
|
||||
stmt = (
|
||||
pg_insert(Relation)
|
||||
.values(values)
|
||||
.on_conflict_do_nothing()
|
||||
.returning(Relation.id)
|
||||
)
|
||||
result = await session.execute(stmt)
|
||||
return len(result.fetchall())
|
||||
else:
|
||||
# SQLite: rowcount works correctly
|
||||
stmt = sqlite_insert(Relation).values(values)
|
||||
stmt = stmt.on_conflict_do_nothing()
|
||||
result = await session.execute(stmt)
|
||||
return result.rowcount if result.rowcount > 0 else 0
|
||||
|
||||
def get_load_options(self) -> List[LoaderOption]:
|
||||
return [selectinload(Relation.from_entity), selectinload(Relation.to_entity)]
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Type, Optional, Any, Sequence, TypeVar, List, Dict
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import (
|
||||
select,
|
||||
|
||||
@@ -4,6 +4,7 @@ from abc import ABC, abstractmethod
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import Executable, Result, text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
|
||||
@@ -5,6 +5,7 @@ import re
|
||||
from datetime import datetime
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
|
||||
@@ -183,7 +183,7 @@ ObservationStr = Annotated[
|
||||
str,
|
||||
BeforeValidator(str.strip), # Clean whitespace
|
||||
MinLen(1), # Ensure non-empty after stripping
|
||||
MaxLen(1000), # Keep reasonable length
|
||||
# No MaxLen - matches DB Text column which has no length restriction
|
||||
]
|
||||
|
||||
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
"""V2 API schemas - ID-based entity references."""
|
||||
"""V2 API schemas - ID-based entity and project references."""
|
||||
|
||||
from basic_memory.schemas.v2.entity import (
|
||||
EntityResolveRequest,
|
||||
EntityResolveResponse,
|
||||
EntityResponseV2,
|
||||
MoveEntityRequestV2,
|
||||
ProjectResolveRequest,
|
||||
ProjectResolveResponse,
|
||||
)
|
||||
from basic_memory.schemas.v2.resource import (
|
||||
CreateResourceRequest,
|
||||
@@ -17,6 +19,8 @@ __all__ = [
|
||||
"EntityResolveResponse",
|
||||
"EntityResponseV2",
|
||||
"MoveEntityRequestV2",
|
||||
"ProjectResolveRequest",
|
||||
"ProjectResolveResponse",
|
||||
"CreateResourceRequest",
|
||||
"UpdateResourceRequest",
|
||||
"ResourceResponse",
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
"""V2 entity schemas with ID-first design."""
|
||||
"""V2 entity and project schemas with ID-first design."""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Literal, Optional
|
||||
@@ -94,3 +94,36 @@ class EntityResponseV2(BaseModel):
|
||||
)
|
||||
|
||||
model_config = ConfigDict(from_attributes=True)
|
||||
|
||||
|
||||
class ProjectResolveRequest(BaseModel):
|
||||
"""Request to resolve a project identifier to a project ID.
|
||||
|
||||
Supports resolution of:
|
||||
- Project names (e.g., "my-project")
|
||||
- Permalinks (e.g., "my-project")
|
||||
"""
|
||||
|
||||
identifier: str = Field(
|
||||
...,
|
||||
description="Project identifier to resolve (name or permalink)",
|
||||
min_length=1,
|
||||
max_length=255,
|
||||
)
|
||||
|
||||
|
||||
class ProjectResolveResponse(BaseModel):
|
||||
"""Response from project identifier resolution.
|
||||
|
||||
Returns the project ID and associated metadata for the resolved project.
|
||||
"""
|
||||
|
||||
project_id: int = Field(..., description="Numeric project ID (primary identifier)")
|
||||
name: str = Field(..., description="Project name")
|
||||
permalink: str = Field(..., description="Project permalink")
|
||||
path: str = Field(..., description="Project file path")
|
||||
is_active: bool = Field(..., description="Whether the project is active")
|
||||
is_default: bool = Field(..., description="Whether the project is the default")
|
||||
resolution_method: Literal["id", "name", "permalink"] = Field(
|
||||
..., description="How the identifier was resolved"
|
||||
)
|
||||
|
||||
@@ -4,6 +4,7 @@ from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
|
||||
@@ -3,8 +3,10 @@
|
||||
import fnmatch
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
|
||||
|
||||
from basic_memory.models import Entity
|
||||
from basic_memory.repository import EntityRepository
|
||||
from basic_memory.schemas.directory import DirectoryNode
|
||||
@@ -12,6 +14,17 @@ from basic_memory.schemas.directory import DirectoryNode
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: Entity) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
class DirectoryService:
|
||||
"""Service for working with directory trees."""
|
||||
|
||||
@@ -77,7 +90,7 @@ class DirectoryService:
|
||||
entity_id=file.id,
|
||||
entity_type=file.entity_type,
|
||||
content_type=file.content_type,
|
||||
updated_at=file.updated_at,
|
||||
updated_at=_mtime_to_datetime(file),
|
||||
)
|
||||
|
||||
# Add to parent directory's children
|
||||
@@ -241,7 +254,7 @@ class DirectoryService:
|
||||
entity_id=file.id,
|
||||
entity_type=file.entity_type,
|
||||
content_type=file.content_type,
|
||||
updated_at=file.updated_at,
|
||||
updated_at=_mtime_to_datetime(file),
|
||||
)
|
||||
|
||||
# Add to parent directory's children
|
||||
|
||||
@@ -8,6 +8,7 @@ import yaml
|
||||
from loguru import logger
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
|
||||
|
||||
from basic_memory.config import ProjectConfig, BasicMemoryConfig
|
||||
from basic_memory.file_utils import (
|
||||
has_frontmatter,
|
||||
@@ -28,6 +29,7 @@ from basic_memory.schemas.base import Permalink
|
||||
from basic_memory.services import BaseService, FileService
|
||||
from basic_memory.services.exceptions import EntityCreationError, EntityNotFoundError
|
||||
from basic_memory.services.link_resolver import LinkResolver
|
||||
from basic_memory.services.search_service import SearchService
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
|
||||
@@ -42,6 +44,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
relation_repository: RelationRepository,
|
||||
file_service: FileService,
|
||||
link_resolver: LinkResolver,
|
||||
search_service: Optional[SearchService] = None,
|
||||
app_config: Optional[BasicMemoryConfig] = None,
|
||||
):
|
||||
super().__init__(entity_repository)
|
||||
@@ -50,6 +53,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
self.entity_parser = entity_parser
|
||||
self.file_service = file_service
|
||||
self.link_resolver = link_resolver
|
||||
self.search_service = search_service
|
||||
self.app_config = app_config
|
||||
|
||||
async def detect_file_path_conflicts(
|
||||
@@ -106,6 +110,9 @@ class EntityService(BaseService[EntityModel]):
|
||||
4. Generate new unique permalink from file path
|
||||
|
||||
Enhanced to detect and handle character-related conflicts.
|
||||
|
||||
Note: Uses lightweight repository methods that skip eager loading of
|
||||
observations and relations for better performance during bulk operations.
|
||||
"""
|
||||
file_path_str = Path(file_path).as_posix()
|
||||
|
||||
@@ -122,16 +129,20 @@ class EntityService(BaseService[EntityModel]):
|
||||
# If markdown has explicit permalink, try to validate it
|
||||
if markdown and markdown.frontmatter.permalink:
|
||||
desired_permalink = markdown.frontmatter.permalink
|
||||
existing = await self.repository.get_by_permalink(desired_permalink)
|
||||
# Use lightweight method - we only need to check file_path
|
||||
existing_file_path = await self.repository.get_file_path_for_permalink(
|
||||
desired_permalink
|
||||
)
|
||||
|
||||
# If no conflict or it's our own file, use as is
|
||||
if not existing or existing.file_path == file_path_str:
|
||||
if not existing_file_path or existing_file_path == file_path_str:
|
||||
return desired_permalink
|
||||
|
||||
# For existing files, try to find current permalink
|
||||
existing = await self.repository.get_by_file_path(file_path_str)
|
||||
if existing:
|
||||
return existing.permalink
|
||||
# Use lightweight method - we only need the permalink
|
||||
existing_permalink = await self.repository.get_permalink_for_file_path(file_path_str)
|
||||
if existing_permalink:
|
||||
return existing_permalink
|
||||
|
||||
# New file - generate permalink
|
||||
if markdown and markdown.frontmatter.permalink:
|
||||
@@ -140,9 +151,10 @@ class EntityService(BaseService[EntityModel]):
|
||||
desired_permalink = generate_permalink(file_path_str)
|
||||
|
||||
# Make unique if needed - enhanced to handle character conflicts
|
||||
# Use lightweight existence check instead of loading full entity
|
||||
permalink = desired_permalink
|
||||
suffix = 1
|
||||
while await self.repository.get_by_permalink(permalink):
|
||||
while await self.repository.permalink_exists(permalink):
|
||||
permalink = f"{desired_permalink}-{suffix}"
|
||||
suffix += 1
|
||||
logger.debug(f"creating unique permalink: {permalink}")
|
||||
@@ -224,8 +236,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
final_content = dump_frontmatter(post)
|
||||
checksum = await self.file_service.write_file(file_path, final_content)
|
||||
|
||||
# parse entity from file
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# parse entity from content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=final_content,
|
||||
)
|
||||
|
||||
# create entity
|
||||
created = await self.create_entity_from_markdown(file_path, entity_markdown)
|
||||
@@ -245,8 +260,12 @@ class EntityService(BaseService[EntityModel]):
|
||||
# Convert file path string to Path
|
||||
file_path = Path(entity.file_path)
|
||||
|
||||
# Read existing frontmatter from the file if it exists
|
||||
existing_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# Read existing content via file_service (for cloud compatibility)
|
||||
existing_content = await self.file_service.read_file_content(file_path)
|
||||
existing_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=existing_content,
|
||||
)
|
||||
|
||||
# Parse content frontmatter to check for user-specified permalink and entity_type
|
||||
content_markdown = None
|
||||
@@ -302,8 +321,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
final_content = dump_frontmatter(merged_post)
|
||||
checksum = await self.file_service.write_file(file_path, final_content)
|
||||
|
||||
# parse entity from file
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# parse entity from content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=final_content,
|
||||
)
|
||||
|
||||
# update entity in db
|
||||
entity = await self.update_entity_and_observations(file_path, entity_markdown)
|
||||
@@ -335,7 +357,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
)
|
||||
entity = entities[0]
|
||||
|
||||
# Delete file first
|
||||
# Delete from search index first (if search_service is available)
|
||||
if self.search_service:
|
||||
await self.search_service.handle_delete(entity)
|
||||
|
||||
# Delete file
|
||||
await self.file_service.delete_entity_file(entity)
|
||||
|
||||
# Delete from DB (this will cascade to observations/relations)
|
||||
@@ -378,7 +404,9 @@ class EntityService(BaseService[EntityModel]):
|
||||
Uses UPSERT approach to handle permalink/file_path conflicts cleanly.
|
||||
"""
|
||||
logger.debug(f"Creating entity: {markdown.frontmatter.title} file_path: {file_path}")
|
||||
model = entity_model_from_markdown(file_path, markdown)
|
||||
model = entity_model_from_markdown(
|
||||
file_path, markdown, project_id=self.repository.project_id
|
||||
)
|
||||
|
||||
# Mark as incomplete because we still need to add relations
|
||||
model.checksum = None
|
||||
@@ -408,6 +436,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
# add new observations
|
||||
observations = [
|
||||
Observation(
|
||||
project_id=self.observation_repository.project_id,
|
||||
entity_id=db_entity.id,
|
||||
content=obs.content,
|
||||
category=obs.category,
|
||||
@@ -474,6 +503,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
|
||||
# Create the relation
|
||||
relation = Relation(
|
||||
project_id=self.relation_repository.project_id,
|
||||
from_id=db_entity.id,
|
||||
to_id=target_id,
|
||||
to_name=target_name,
|
||||
@@ -546,8 +576,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
# Write the updated content back to the file
|
||||
checksum = await self.file_service.write_file(file_path, new_content)
|
||||
|
||||
# Parse the updated file to get new observations/relations
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# Parse the content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=new_content,
|
||||
)
|
||||
|
||||
# Update entity and its relationships
|
||||
entity = await self.update_entity_and_observations(file_path, entity_markdown)
|
||||
@@ -766,23 +799,20 @@ class EntityService(BaseService[EntityModel]):
|
||||
raise ValueError(f"Invalid destination path: {destination_path}")
|
||||
|
||||
# 3. Validate paths
|
||||
source_file = project_config.home / current_path
|
||||
destination_file = project_config.home / destination_path
|
||||
|
||||
# Validate source exists
|
||||
if not source_file.exists():
|
||||
# NOTE: In tenantless/cloud mode, we cannot rely on local filesystem paths.
|
||||
# Use FileService for existence checks and moving.
|
||||
if not await self.file_service.exists(current_path):
|
||||
raise ValueError(f"Source file not found: {current_path}")
|
||||
|
||||
# Check if destination already exists
|
||||
if destination_file.exists():
|
||||
if await self.file_service.exists(destination_path):
|
||||
raise ValueError(f"Destination already exists: {destination_path}")
|
||||
|
||||
try:
|
||||
# 4. Create destination directory if needed
|
||||
destination_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
# 4. Ensure destination directory if needed (no-op for S3)
|
||||
await self.file_service.ensure_directory(Path(destination_path).parent)
|
||||
|
||||
# 5. Move physical file
|
||||
source_file.rename(destination_file)
|
||||
# 5. Move physical file via FileService (filesystem rename or cloud move)
|
||||
await self.file_service.move_file(current_path, destination_path)
|
||||
logger.info(f"Moved file: {current_path} -> {destination_path}")
|
||||
|
||||
# 6. Prepare database updates
|
||||
@@ -821,12 +851,14 @@ class EntityService(BaseService[EntityModel]):
|
||||
|
||||
except Exception as e:
|
||||
# Rollback: try to restore original file location if move succeeded
|
||||
if destination_file.exists() and not source_file.exists():
|
||||
try:
|
||||
destination_file.rename(source_file)
|
||||
try:
|
||||
if await self.file_service.exists(
|
||||
destination_path
|
||||
) and not await self.file_service.exists(current_path):
|
||||
await self.file_service.move_file(destination_path, current_path)
|
||||
logger.info(f"Rolled back file move: {destination_path} -> {current_path}")
|
||||
except Exception as rollback_error: # pragma: no cover
|
||||
logger.error(f"Failed to rollback file move: {rollback_error}")
|
||||
except Exception as rollback_error: # pragma: no cover
|
||||
logger.error(f"Failed to rollback file move: {rollback_error}")
|
||||
|
||||
# Re-raise the original error with context
|
||||
raise ValueError(f"Move failed: {str(e)}") from e
|
||||
|
||||
@@ -3,15 +3,19 @@
|
||||
import asyncio
|
||||
import hashlib
|
||||
import mimetypes
|
||||
from os import stat_result
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Tuple, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, Union
|
||||
|
||||
import aiofiles
|
||||
|
||||
import yaml
|
||||
|
||||
from basic_memory import file_utils
|
||||
from basic_memory.file_utils import FileError, ParseError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from basic_memory.config import BasicMemoryConfig
|
||||
from basic_memory.file_utils import FileError, FileMetadata, ParseError
|
||||
from basic_memory.markdown.markdown_processor import MarkdownProcessor
|
||||
from basic_memory.models import Entity as EntityModel
|
||||
from basic_memory.schemas import Entity as EntitySchema
|
||||
@@ -41,9 +45,11 @@ class FileService:
|
||||
base_path: Path,
|
||||
markdown_processor: MarkdownProcessor,
|
||||
max_concurrent_files: int = 10,
|
||||
app_config: Optional["BasicMemoryConfig"] = None,
|
||||
):
|
||||
self.base_path = base_path.resolve() # Get absolute path
|
||||
self.markdown_processor = markdown_processor
|
||||
self.app_config = app_config
|
||||
# Semaphore to limit concurrent file operations
|
||||
# Prevents OOM on large projects by processing files in batches
|
||||
self._file_semaphore = asyncio.Semaphore(max_concurrent_files)
|
||||
@@ -148,12 +154,15 @@ class FileService:
|
||||
Handles both absolute and relative paths. Relative paths are resolved
|
||||
against base_path.
|
||||
|
||||
If format_on_save is enabled in config, runs the configured formatter
|
||||
after writing and returns the checksum of the formatted content.
|
||||
|
||||
Args:
|
||||
path: Where to write (Path or string)
|
||||
content: Content to write
|
||||
|
||||
Returns:
|
||||
Checksum of written content
|
||||
Checksum of written content (or formatted content if formatting enabled)
|
||||
|
||||
Raises:
|
||||
FileOperationError: If write fails
|
||||
@@ -176,8 +185,17 @@ class FileService:
|
||||
|
||||
await file_utils.write_file_atomic(full_path, content)
|
||||
|
||||
# Compute and return checksum
|
||||
checksum = await file_utils.compute_checksum(content)
|
||||
# Format file if configured
|
||||
final_content = content
|
||||
if self.app_config:
|
||||
formatted_content = await file_utils.format_file(
|
||||
full_path, self.app_config, is_markdown=self.is_markdown(path)
|
||||
)
|
||||
if formatted_content is not None:
|
||||
final_content = formatted_content
|
||||
|
||||
# Compute and return checksum of final content
|
||||
checksum = await file_utils.compute_checksum(final_content)
|
||||
logger.debug(f"File write completed path={full_path}, {checksum=}")
|
||||
return checksum
|
||||
|
||||
@@ -220,6 +238,41 @@ class FileService:
|
||||
logger.exception("File read error", path=str(full_path), error=str(e))
|
||||
raise FileOperationError(f"Failed to read file: {e}")
|
||||
|
||||
async def read_file_bytes(self, path: FilePath) -> bytes:
|
||||
"""Read file content as bytes using true async I/O with aiofiles.
|
||||
|
||||
This method reads files in binary mode, suitable for non-text files
|
||||
like images, PDFs, etc. For cloud compatibility with S3FileService.
|
||||
|
||||
Args:
|
||||
path: Path to read (Path or string)
|
||||
|
||||
Returns:
|
||||
File content as bytes
|
||||
|
||||
Raises:
|
||||
FileOperationError: If read fails
|
||||
"""
|
||||
# Convert string to Path if needed
|
||||
path_obj = self.base_path / path if isinstance(path, str) else path
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
|
||||
try:
|
||||
logger.debug("Reading file bytes", operation="read_file_bytes", path=str(full_path))
|
||||
async with aiofiles.open(full_path, mode="rb") as f:
|
||||
content = await f.read()
|
||||
|
||||
logger.debug(
|
||||
"File read completed",
|
||||
path=str(full_path),
|
||||
content_length=len(content),
|
||||
)
|
||||
return content
|
||||
|
||||
except Exception as e:
|
||||
logger.exception("File read error", path=str(full_path), error=str(e))
|
||||
raise FileOperationError(f"Failed to read file: {e}")
|
||||
|
||||
async def read_file(self, path: FilePath) -> Tuple[str, str]:
|
||||
"""Read file and compute checksum using true async I/O.
|
||||
|
||||
@@ -276,6 +329,43 @@ class FileService:
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
full_path.unlink(missing_ok=True)
|
||||
|
||||
async def move_file(self, source: FilePath, destination: FilePath) -> None:
|
||||
"""Move/rename a file from source to destination.
|
||||
|
||||
This method abstracts the underlying storage (filesystem vs cloud).
|
||||
Default implementation uses atomic filesystem rename, but cloud-backed
|
||||
implementations (e.g., S3) can override to copy+delete.
|
||||
|
||||
Args:
|
||||
source: Source path (relative to base_path or absolute)
|
||||
destination: Destination path (relative to base_path or absolute)
|
||||
|
||||
Raises:
|
||||
FileOperationError: If the move fails
|
||||
"""
|
||||
# Convert strings to Paths and resolve relative paths against base_path
|
||||
src_obj = self.base_path / source if isinstance(source, str) else source
|
||||
dst_obj = self.base_path / destination if isinstance(destination, str) else destination
|
||||
src_full = src_obj if src_obj.is_absolute() else self.base_path / src_obj
|
||||
dst_full = dst_obj if dst_obj.is_absolute() else self.base_path / dst_obj
|
||||
|
||||
try:
|
||||
# Ensure destination directory exists
|
||||
await self.ensure_directory(dst_full.parent)
|
||||
|
||||
# Use semaphore for concurrency control and run blocking rename in executor
|
||||
async with self._file_semaphore:
|
||||
loop = asyncio.get_event_loop()
|
||||
await loop.run_in_executor(None, lambda: src_full.rename(dst_full))
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
"File move error",
|
||||
source=str(src_full),
|
||||
destination=str(dst_full),
|
||||
error=str(e),
|
||||
)
|
||||
raise FileOperationError(f"Failed to move file {source} -> {destination}: {e}")
|
||||
|
||||
async def update_frontmatter(self, path: FilePath, updates: Dict[str, Any]) -> str:
|
||||
"""Update frontmatter fields in a file while preserving all content.
|
||||
|
||||
@@ -332,7 +422,17 @@ class FileService:
|
||||
)
|
||||
|
||||
await file_utils.write_file_atomic(full_path, final_content)
|
||||
return await file_utils.compute_checksum(final_content)
|
||||
|
||||
# Format file if configured
|
||||
content_for_checksum = final_content
|
||||
if self.app_config:
|
||||
formatted_content = await file_utils.format_file(
|
||||
full_path, self.app_config, is_markdown=self.is_markdown(path)
|
||||
)
|
||||
if formatted_content is not None:
|
||||
content_for_checksum = formatted_content
|
||||
|
||||
return await file_utils.compute_checksum(content_for_checksum)
|
||||
|
||||
except Exception as e:
|
||||
# Only log real errors (not YAML parsing, which is handled above)
|
||||
@@ -381,20 +481,31 @@ class FileService:
|
||||
logger.error("Failed to compute checksum", path=str(full_path), error=str(e))
|
||||
raise FileError(f"Failed to compute checksum for {path}: {e}")
|
||||
|
||||
def file_stats(self, path: FilePath) -> stat_result:
|
||||
"""Return file stats for a given path.
|
||||
async def get_file_metadata(self, path: FilePath) -> FileMetadata:
|
||||
"""Return file metadata for a given path.
|
||||
|
||||
This method is async to support cloud implementations (S3FileService)
|
||||
where file metadata requires async operations (head_object).
|
||||
|
||||
Args:
|
||||
path: Path to the file (Path or string)
|
||||
|
||||
Returns:
|
||||
File statistics
|
||||
FileMetadata with size, created_at, and modified_at
|
||||
"""
|
||||
# Convert string to Path if needed
|
||||
path_obj = self.base_path / path if isinstance(path, str) else path
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
# get file timestamps
|
||||
return full_path.stat()
|
||||
|
||||
# Run blocking stat() in thread pool to maintain async compatibility
|
||||
loop = asyncio.get_event_loop()
|
||||
stat_result = await loop.run_in_executor(None, full_path.stat)
|
||||
|
||||
return FileMetadata(
|
||||
size=stat_result.st_size,
|
||||
created_at=datetime.fromtimestamp(stat_result.st_ctime).astimezone(),
|
||||
modified_at=datetime.fromtimestamp(stat_result.st_mtime).astimezone(),
|
||||
)
|
||||
|
||||
def content_type(self, path: FilePath) -> str:
|
||||
"""Return content_type for a given path.
|
||||
|
||||
@@ -5,8 +5,11 @@ to ensure consistent application startup across all entry points.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory import db
|
||||
@@ -27,15 +30,12 @@ async def initialize_database(app_config: BasicMemoryConfig) -> None:
|
||||
Database migrations are now handled automatically when the database
|
||||
connection is first established via get_or_create_db().
|
||||
"""
|
||||
# Trigger database initialization and migrations by getting the database connection
|
||||
try:
|
||||
await db.get_or_create_db(app_config.database_path)
|
||||
logger.info("Database initialization completed")
|
||||
except Exception as e:
|
||||
logger.error(f"Error initializing database: {e}")
|
||||
# Allow application to continue - it might still work
|
||||
# depending on what the error was, and will fail with a
|
||||
# more specific error if the database is actually unusable
|
||||
logger.error(f"Error during database initialization: {e}")
|
||||
raise
|
||||
|
||||
|
||||
async def reconcile_projects_with_config(app_config: BasicMemoryConfig):
|
||||
@@ -49,31 +49,29 @@ async def reconcile_projects_with_config(app_config: BasicMemoryConfig):
|
||||
"""
|
||||
logger.info("Reconciling projects from config with database...")
|
||||
|
||||
# Get database session - migrations handled centrally
|
||||
# Get database session (engine already created by initialize_database)
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
ensure_migrations=False,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
# Import ProjectService here to avoid circular imports
|
||||
from basic_memory.services.project_service import ProjectService
|
||||
|
||||
# Create project service and synchronize projects
|
||||
project_service = ProjectService(repository=project_repository)
|
||||
try:
|
||||
# Create project service and synchronize projects
|
||||
project_service = ProjectService(repository=project_repository)
|
||||
await project_service.synchronize_projects()
|
||||
logger.info("Projects successfully reconciled between config and database")
|
||||
except Exception as e:
|
||||
# Log the error but continue with initialization
|
||||
logger.error(f"Error during project synchronization: {e}")
|
||||
logger.info("Continuing with initialization despite synchronization error")
|
||||
|
||||
|
||||
async def initialize_file_sync(
|
||||
app_config: BasicMemoryConfig,
|
||||
):
|
||||
) -> None:
|
||||
"""Initialize file synchronization services. This function starts the watch service and does not return
|
||||
|
||||
Args:
|
||||
@@ -82,15 +80,20 @@ async def initialize_file_sync(
|
||||
Returns:
|
||||
The watch service task that's monitoring file changes
|
||||
"""
|
||||
# Never start file watching during tests. Even "background" watchers add tasks/threads
|
||||
# and can interact badly with strict asyncio teardown (especially on Windows/aiosqlite).
|
||||
# Skip file sync in test environments to avoid interference with tests
|
||||
if app_config.is_test_env:
|
||||
logger.info("Test environment detected - skipping file sync initialization")
|
||||
return None
|
||||
|
||||
# delay import
|
||||
from basic_memory.sync import WatchService
|
||||
|
||||
# Load app configuration - migrations handled centrally
|
||||
# Get database session (migrations already run if needed)
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
ensure_migrations=False,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
@@ -104,6 +107,12 @@ async def initialize_file_sync(
|
||||
# Get active projects
|
||||
active_projects = await project_repository.get_active_projects()
|
||||
|
||||
# Filter to constrained project if MCP server was started with --project
|
||||
constrained_project = os.environ.get("BASIC_MEMORY_MCP_PROJECT")
|
||||
if constrained_project:
|
||||
active_projects = [p for p in active_projects if p.name == constrained_project]
|
||||
logger.info(f"Background sync constrained to project: {constrained_project}")
|
||||
|
||||
# Start sync for all projects as background tasks (non-blocking)
|
||||
async def sync_project_background(project: Project):
|
||||
"""Sync a single project in the background."""
|
||||
@@ -131,12 +140,10 @@ async def initialize_file_sync(
|
||||
|
||||
# Then start the watch service in the background
|
||||
logger.info("Starting watch service for all projects")
|
||||
|
||||
# run the watch service
|
||||
try:
|
||||
await watch_service.run()
|
||||
logger.info("Watch service started")
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error starting watch service: {e}")
|
||||
await watch_service.run()
|
||||
logger.info("Watch service started")
|
||||
|
||||
return None
|
||||
|
||||
@@ -155,6 +162,11 @@ async def initialize_app(
|
||||
Args:
|
||||
app_config: The Basic Memory project configuration
|
||||
"""
|
||||
# Skip initialization in cloud mode - cloud manages its own projects
|
||||
if app_config.cloud_mode_enabled:
|
||||
logger.debug("Skipping initialization in cloud mode - projects managed by cloud")
|
||||
return
|
||||
|
||||
logger.info("Initializing app...")
|
||||
# Initialize database first
|
||||
await initialize_database(app_config)
|
||||
@@ -181,11 +193,24 @@ def ensure_initialization(app_config: BasicMemoryConfig) -> None:
|
||||
logger.debug("Skipping initialization in cloud mode - projects managed by cloud")
|
||||
return
|
||||
|
||||
try:
|
||||
result = asyncio.run(initialize_app(app_config))
|
||||
logger.info(f"Initialization completed successfully: result={result}")
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.exception(f"Error during initialization: {e}")
|
||||
# Continue execution even if initialization fails
|
||||
# The command might still work, or will fail with a
|
||||
# more specific error message
|
||||
async def _init_and_cleanup():
|
||||
"""Initialize app and clean up database connections.
|
||||
|
||||
Database connections created during initialization must be cleaned up
|
||||
before the event loop closes, otherwise the process will hang indefinitely.
|
||||
"""
|
||||
try:
|
||||
await initialize_app(app_config)
|
||||
finally:
|
||||
# Always cleanup database connections to prevent process hang
|
||||
await db.shutdown_db()
|
||||
|
||||
# On Windows, use SelectorEventLoop to avoid ProactorEventLoop cleanup issues
|
||||
# The ProactorEventLoop can raise "IndexError: pop from an empty deque" during
|
||||
# event loop cleanup when there are pending handles. SelectorEventLoop is more
|
||||
# stable for our use case (no subprocess pipes or named pipes needed).
|
||||
if sys.platform == "win32":
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
|
||||
asyncio.run(_init_and_cleanup())
|
||||
logger.info("Initialization completed successfully")
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Optional, Tuple
|
||||
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.models import Entity
|
||||
|
||||
@@ -8,6 +8,7 @@ from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, Sequence
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
@@ -23,9 +24,6 @@ from basic_memory.config import WATCH_STATUS_JSON, ConfigManager, get_project_co
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
|
||||
config = ConfigManager().config
|
||||
|
||||
|
||||
class ProjectService:
|
||||
"""Service for managing Basic Memory projects."""
|
||||
|
||||
@@ -143,6 +141,7 @@ class ProjectService:
|
||||
"""
|
||||
# If project_root is set, constrain all projects to that directory
|
||||
project_root = self.config_manager.config.project_root
|
||||
sanitized_name = None
|
||||
if project_root:
|
||||
base_path = Path(project_root)
|
||||
|
||||
@@ -199,14 +198,15 @@ class ProjectService:
|
||||
f"Projects cannot share directory trees."
|
||||
)
|
||||
|
||||
# First add to config file (this will validate the project doesn't exist)
|
||||
project_config = self.config_manager.add_project(name, resolved_path)
|
||||
if not self.config_manager.config.cloud_mode:
|
||||
# First add to config file (this will validate the project doesn't exist)
|
||||
self.config_manager.add_project(name, resolved_path)
|
||||
|
||||
# Then add to database
|
||||
project_data = {
|
||||
"name": name,
|
||||
"path": resolved_path,
|
||||
"permalink": generate_permalink(project_config.name),
|
||||
"permalink": sanitized_name,
|
||||
"is_active": True,
|
||||
# Don't set is_default=False to avoid UNIQUE constraint issues
|
||||
# Let it default to NULL, only set to True when explicitly making default
|
||||
|
||||
@@ -4,6 +4,7 @@ import ast
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Set
|
||||
|
||||
|
||||
from dateparser import parse
|
||||
from fastapi import BackgroundTasks
|
||||
from loguru import logger
|
||||
@@ -15,6 +16,21 @@ from basic_memory.repository.search_repository import SearchRepository, SearchIn
|
||||
from basic_memory.schemas.search import SearchQuery, SearchItemType
|
||||
from basic_memory.services import FileService
|
||||
|
||||
# Maximum size for content_stems field to stay under Postgres's 8KB index row limit.
|
||||
# We use 6000 characters to leave headroom for other indexed columns and overhead.
|
||||
MAX_CONTENT_STEMS_SIZE = 6000
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: Entity) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
class SearchService:
|
||||
"""Service for search operations.
|
||||
@@ -193,7 +209,7 @@ class SearchService:
|
||||
"entity_type": entity.entity_type,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
@@ -256,6 +272,10 @@ class SearchService:
|
||||
|
||||
entity_content_stems = "\n".join(p for p in content_stems if p and p.strip())
|
||||
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(entity_content_stems) > MAX_CONTENT_STEMS_SIZE:
|
||||
entity_content_stems = entity_content_stems[:MAX_CONTENT_STEMS_SIZE]
|
||||
|
||||
# Add entity row
|
||||
rows_to_index.append(
|
||||
SearchIndexRow(
|
||||
@@ -271,17 +291,28 @@ class SearchService:
|
||||
"entity_type": entity.entity_type,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
|
||||
# Add observation rows
|
||||
# Add observation rows - dedupe by permalink to avoid unique constraint violations
|
||||
# Two observations with same entity/category/content generate identical permalinks
|
||||
seen_permalinks: set[str] = {entity.permalink} if entity.permalink else set()
|
||||
for obs in entity.observations:
|
||||
obs_permalink = obs.permalink
|
||||
if obs_permalink in seen_permalinks:
|
||||
logger.debug(f"Skipping duplicate observation permalink: {obs_permalink}")
|
||||
continue
|
||||
seen_permalinks.add(obs_permalink)
|
||||
|
||||
# Index with parent entity's file path since that's where it's defined
|
||||
obs_content_stems = "\n".join(
|
||||
p for p in self._generate_variants(obs.content) if p and p.strip()
|
||||
)
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(obs_content_stems) > MAX_CONTENT_STEMS_SIZE:
|
||||
obs_content_stems = obs_content_stems[:MAX_CONTENT_STEMS_SIZE]
|
||||
rows_to_index.append(
|
||||
SearchIndexRow(
|
||||
id=obs.id,
|
||||
@@ -289,7 +320,7 @@ class SearchService:
|
||||
title=f"{obs.category}: {obs.content[:100]}...",
|
||||
content_stems=obs_content_stems,
|
||||
content_snippet=obs.content,
|
||||
permalink=obs.permalink,
|
||||
permalink=obs_permalink,
|
||||
file_path=entity.file_path,
|
||||
category=obs.category,
|
||||
entity_id=entity.id,
|
||||
@@ -297,7 +328,7 @@ class SearchService:
|
||||
"tags": obs.tags,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
@@ -327,7 +358,7 @@ class SearchService:
|
||||
to_id=rel.to_id,
|
||||
relation_type=rel.relation_type,
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from collections import OrderedDict
|
||||
from dataclasses import dataclass, field
|
||||
@@ -10,7 +11,7 @@ from pathlib import Path
|
||||
from typing import AsyncIterator, Dict, List, Optional, Set, Tuple
|
||||
|
||||
import aiofiles.os
|
||||
import logfire
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
|
||||
@@ -215,17 +216,12 @@ class SyncService:
|
||||
f"path={path}, error={error}"
|
||||
)
|
||||
|
||||
# Record metric for file failure
|
||||
logfire.metric_counter("sync.circuit_breaker.failures").add(1)
|
||||
|
||||
# Log when threshold is reached
|
||||
if failure_info.count >= MAX_CONSECUTIVE_FAILURES:
|
||||
logger.error(
|
||||
f"File {path} has failed {MAX_CONSECUTIVE_FAILURES} times and will be skipped. "
|
||||
f"First failure: {failure_info.first_failure}, Last error: {error}"
|
||||
)
|
||||
# Record metric for file being blocked by circuit breaker
|
||||
logfire.metric_counter("sync.circuit_breaker.blocked_files").add(1)
|
||||
else:
|
||||
# Create new failure record
|
||||
self._file_failures[path] = FileFailureInfo(
|
||||
@@ -255,7 +251,6 @@ class SyncService:
|
||||
logger.info(f"Clearing failure history for {path} after successful sync")
|
||||
del self._file_failures[path]
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync(
|
||||
self, directory: Path, project_name: Optional[str] = None, force_full: bool = False
|
||||
) -> SyncReport:
|
||||
@@ -282,63 +277,58 @@ class SyncService:
|
||||
)
|
||||
|
||||
# sync moves first
|
||||
with logfire.span("process_moves", move_count=len(report.moves)):
|
||||
for old_path, new_path in report.moves.items():
|
||||
# in the case where a file has been deleted and replaced by another file
|
||||
# it will show up in the move and modified lists, so handle it in modified
|
||||
if new_path in report.modified:
|
||||
report.modified.remove(new_path)
|
||||
logger.debug(
|
||||
f"File marked as moved and modified: old_path={old_path}, new_path={new_path}"
|
||||
)
|
||||
else:
|
||||
await self.handle_move(old_path, new_path)
|
||||
for old_path, new_path in report.moves.items():
|
||||
# in the case where a file has been deleted and replaced by another file
|
||||
# it will show up in the move and modified lists, so handle it in modified
|
||||
if new_path in report.modified:
|
||||
report.modified.remove(new_path)
|
||||
logger.debug(
|
||||
f"File marked as moved and modified: old_path={old_path}, new_path={new_path}"
|
||||
)
|
||||
else:
|
||||
await self.handle_move(old_path, new_path)
|
||||
|
||||
# deleted next
|
||||
with logfire.span("process_deletes", delete_count=len(report.deleted)):
|
||||
for path in report.deleted:
|
||||
await self.handle_delete(path)
|
||||
for path in report.deleted:
|
||||
await self.handle_delete(path)
|
||||
|
||||
# then new and modified
|
||||
with logfire.span("process_new_files", new_count=len(report.new)):
|
||||
for path in report.new:
|
||||
entity, _ = await self.sync_file(path, new=True)
|
||||
for path in report.new:
|
||||
entity, _ = await self.sync_file(path, new=True)
|
||||
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
)
|
||||
|
||||
with logfire.span("process_modified_files", modified_count=len(report.modified)):
|
||||
for path in report.modified:
|
||||
entity, _ = await self.sync_file(path, new=False)
|
||||
for path in report.modified:
|
||||
entity, _ = await self.sync_file(path, new=False)
|
||||
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
)
|
||||
|
||||
# Only resolve relations if there were actual changes
|
||||
# If no files changed, no new unresolved relations could have been created
|
||||
with logfire.span("resolve_relations"):
|
||||
if report.total > 0:
|
||||
await self.resolve_relations()
|
||||
else:
|
||||
logger.info("Skipping relation resolution - no file changes detected")
|
||||
if report.total > 0:
|
||||
await self.resolve_relations()
|
||||
else:
|
||||
logger.info("Skipping relation resolution - no file changes detected")
|
||||
|
||||
# Update scan watermark after successful sync
|
||||
# Use the timestamp from sync start (not end) to ensure we catch files
|
||||
@@ -361,15 +351,6 @@ class SyncService:
|
||||
|
||||
duration_ms = int((time.time() - start_time) * 1000)
|
||||
|
||||
# Record metrics for sync operation
|
||||
logfire.metric_histogram("sync.duration", unit="ms").record(duration_ms)
|
||||
logfire.metric_counter("sync.files.new").add(len(report.new))
|
||||
logfire.metric_counter("sync.files.modified").add(len(report.modified))
|
||||
logfire.metric_counter("sync.files.deleted").add(len(report.deleted))
|
||||
logfire.metric_counter("sync.files.moved").add(len(report.moves))
|
||||
if report.skipped_files:
|
||||
logfire.metric_counter("sync.files.skipped").add(len(report.skipped_files))
|
||||
|
||||
# Log summary with skipped files if any
|
||||
if report.skipped_files:
|
||||
logger.warning(
|
||||
@@ -390,7 +371,6 @@ class SyncService:
|
||||
|
||||
return report
|
||||
|
||||
@logfire.instrument()
|
||||
async def scan(self, directory, force_full: bool = False):
|
||||
"""Smart scan using watermark and file count for large project optimization.
|
||||
|
||||
@@ -472,12 +452,6 @@ class SyncService:
|
||||
logger.warning("No scan watermark available, falling back to full scan")
|
||||
file_paths_to_scan = await self._scan_directory_full(directory)
|
||||
|
||||
# Record scan type metric
|
||||
logfire.metric_counter(f"sync.scan.{scan_type}").add(1)
|
||||
logfire.metric_histogram("sync.scan.files_scanned", unit="files").record(
|
||||
len(file_paths_to_scan)
|
||||
)
|
||||
|
||||
# Step 3: Process each file with mtime-based comparison
|
||||
scanned_paths: Set[str] = set()
|
||||
changed_checksums: Dict[str, str] = {}
|
||||
@@ -589,7 +563,6 @@ class SyncService:
|
||||
report.checksums = changed_checksums
|
||||
|
||||
scan_duration_ms = int((time.time() - scan_start_time) * 1000)
|
||||
logfire.metric_histogram("sync.scan.duration", unit="ms").record(scan_duration_ms)
|
||||
|
||||
logger.info(
|
||||
f"Completed {scan_type} scan for directory {directory} in {scan_duration_ms}ms, "
|
||||
@@ -599,7 +572,6 @@ class SyncService:
|
||||
)
|
||||
return report
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_file(
|
||||
self, path: str, new: bool = True
|
||||
) -> Tuple[Optional[Entity], Optional[str]]:
|
||||
@@ -638,6 +610,16 @@ class SyncService:
|
||||
)
|
||||
return entity, checksum
|
||||
|
||||
except FileNotFoundError:
|
||||
# File exists in database but not on filesystem
|
||||
# This indicates a database/filesystem inconsistency - treat as deletion
|
||||
logger.warning(
|
||||
f"File not found during sync, treating as deletion: path={path}. "
|
||||
"This may indicate a race condition or manual file deletion."
|
||||
)
|
||||
await self.handle_delete(path)
|
||||
return None, None
|
||||
|
||||
except Exception as e:
|
||||
# Check if this is a fatal error (or caused by one)
|
||||
# Fatal errors like project deletion should terminate sync immediately
|
||||
@@ -654,7 +636,6 @@ class SyncService:
|
||||
|
||||
return None, None
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_markdown_file(self, path: str, new: bool = True) -> Tuple[Optional[Entity], str]:
|
||||
"""Sync a markdown file with full processing.
|
||||
|
||||
@@ -672,12 +653,19 @@ class SyncService:
|
||||
file_contains_frontmatter = has_frontmatter(file_content)
|
||||
|
||||
# Get file timestamps for tracking modification times
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
created = datetime.fromtimestamp(file_stats.st_ctime).astimezone()
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
created = file_metadata.created_at
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
# entity markdown will always contain front matter, so it can be used up create/update the entity
|
||||
entity_markdown = await self.entity_parser.parse_file(path)
|
||||
# Parse markdown content with file metadata (avoids redundant file read/stat)
|
||||
# This enables cloud implementations (S3FileService) to provide metadata from head_object
|
||||
abs_path = self.file_service.base_path / path
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=abs_path,
|
||||
content=file_content,
|
||||
mtime=file_metadata.modified_at.timestamp(),
|
||||
ctime=file_metadata.created_at.timestamp(),
|
||||
)
|
||||
|
||||
# if the file contains frontmatter, resolve a permalink (unless disabled)
|
||||
if file_contains_frontmatter and not self.app_config.disable_permalinks:
|
||||
@@ -723,8 +711,8 @@ class SyncService:
|
||||
"checksum": final_checksum,
|
||||
"created_at": created,
|
||||
"updated_at": modified,
|
||||
"mtime": file_stats.st_mtime,
|
||||
"size": file_stats.st_size,
|
||||
"mtime": file_metadata.modified_at.timestamp(),
|
||||
"size": file_metadata.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -737,7 +725,6 @@ class SyncService:
|
||||
# Return the final checksum to ensure everything is consistent
|
||||
return entity, final_checksum
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_regular_file(self, path: str, new: bool = True) -> Tuple[Optional[Entity], str]:
|
||||
"""Sync a non-markdown file with basic tracking.
|
||||
|
||||
@@ -754,9 +741,9 @@ class SyncService:
|
||||
await self.entity_service.resolve_permalink(path, skip_conflict_check=True)
|
||||
|
||||
# get file timestamps
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
created = datetime.fromtimestamp(file_stats.st_ctime).astimezone()
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
created = file_metadata.created_at
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
# get mime type
|
||||
content_type = self.file_service.content_type(path)
|
||||
@@ -772,8 +759,8 @@ class SyncService:
|
||||
created_at=created,
|
||||
updated_at=modified,
|
||||
content_type=content_type,
|
||||
mtime=file_stats.st_mtime,
|
||||
size=file_stats.st_size,
|
||||
mtime=file_metadata.modified_at.timestamp(),
|
||||
size=file_metadata.size,
|
||||
)
|
||||
)
|
||||
return entity, checksum
|
||||
@@ -789,15 +776,15 @@ class SyncService:
|
||||
logger.error(f"Entity not found after constraint violation, path={path}")
|
||||
raise ValueError(f"Entity not found after constraint violation: {path}")
|
||||
|
||||
# Re-get file stats since we're in update path
|
||||
file_stats_for_update = self.file_service.file_stats(path)
|
||||
# Re-get file metadata since we're in update path
|
||||
file_metadata_for_update = await self.file_service.get_file_metadata(path)
|
||||
updated = await self.entity_repository.update(
|
||||
entity.id,
|
||||
{
|
||||
"file_path": path,
|
||||
"checksum": checksum,
|
||||
"mtime": file_stats_for_update.st_mtime,
|
||||
"size": file_stats_for_update.st_size,
|
||||
"mtime": file_metadata_for_update.modified_at.timestamp(),
|
||||
"size": file_metadata_for_update.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -811,8 +798,8 @@ class SyncService:
|
||||
raise
|
||||
else:
|
||||
# Get file timestamps for updating modification time
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
entity = await self.entity_repository.get_by_file_path(path)
|
||||
if entity is None: # pragma: no cover
|
||||
@@ -827,8 +814,8 @@ class SyncService:
|
||||
"file_path": path,
|
||||
"checksum": checksum,
|
||||
"updated_at": modified,
|
||||
"mtime": file_stats.st_mtime,
|
||||
"size": file_stats.st_size,
|
||||
"mtime": file_metadata.modified_at.timestamp(),
|
||||
"size": file_metadata.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -838,7 +825,6 @@ class SyncService:
|
||||
|
||||
return updated, checksum
|
||||
|
||||
@logfire.instrument()
|
||||
async def handle_delete(self, file_path: str):
|
||||
"""Handle complete entity deletion including search index cleanup."""
|
||||
|
||||
@@ -870,7 +856,6 @@ class SyncService:
|
||||
else:
|
||||
await self.search_service.delete_by_entity_id(entity.id)
|
||||
|
||||
@logfire.instrument()
|
||||
async def handle_move(self, old_path, new_path):
|
||||
logger.debug("Moving entity", old_path=old_path, new_path=new_path)
|
||||
|
||||
@@ -975,7 +960,6 @@ class SyncService:
|
||||
# update search index
|
||||
await self.search_service.index_entity(updated)
|
||||
|
||||
@logfire.instrument()
|
||||
async def resolve_relations(self, entity_id: int | None = None):
|
||||
"""Try to resolve unresolved relations.
|
||||
|
||||
@@ -1026,16 +1010,27 @@ class SyncService:
|
||||
"to_name": resolved_entity.title,
|
||||
},
|
||||
)
|
||||
except IntegrityError: # pragma: no cover
|
||||
# update search index only on successful resolution
|
||||
await self.search_service.index_entity(resolved_entity)
|
||||
except IntegrityError:
|
||||
# IntegrityError means a relation with this (from_id, to_id, relation_type)
|
||||
# already exists. The UPDATE was rolled back, so our unresolved relation
|
||||
# (to_id=NULL) still exists in the database. We delete it because:
|
||||
# 1. It's redundant - a resolved relation already captures this relationship
|
||||
# 2. If we don't delete it, future syncs will try to resolve it again
|
||||
# and get the same IntegrityError
|
||||
logger.debug(
|
||||
"Ignoring duplicate relation "
|
||||
"Deleting duplicate unresolved relation "
|
||||
f"relation_id={relation.id} "
|
||||
f"from_id={relation.from_id} "
|
||||
f"to_name={relation.to_name}"
|
||||
f"to_name={relation.to_name} "
|
||||
f"resolved_to_id={resolved_entity.id}"
|
||||
)
|
||||
|
||||
# update search index
|
||||
await self.search_service.index_entity(resolved_entity)
|
||||
try:
|
||||
await self.relation_repository.delete(relation.id)
|
||||
except Exception as e:
|
||||
# Log but don't fail - the relation may have been deleted already
|
||||
logger.debug(f"Could not delete duplicate relation {relation.id}: {e}")
|
||||
|
||||
async def _quick_count_files(self, directory: Path) -> int:
|
||||
"""Fast file count using find command.
|
||||
@@ -1043,12 +1038,22 @@ class SyncService:
|
||||
Uses subprocess to leverage OS-level file counting which is much faster
|
||||
than Python iteration, especially on network filesystems like TigrisFS.
|
||||
|
||||
On Windows, subprocess is not supported with SelectorEventLoop (which we use
|
||||
to avoid aiosqlite cleanup issues), so we fall back to Python-based counting.
|
||||
|
||||
Args:
|
||||
directory: Directory to count files in
|
||||
|
||||
Returns:
|
||||
Number of files in directory (recursive)
|
||||
"""
|
||||
# Windows with SelectorEventLoop doesn't support subprocess
|
||||
if sys.platform == "win32":
|
||||
count = 0
|
||||
async for _ in self.scan_directory(directory):
|
||||
count += 1
|
||||
return count
|
||||
|
||||
process = await asyncio.create_subprocess_shell(
|
||||
f'find "{directory}" -type f | wc -l',
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
@@ -1063,8 +1068,6 @@ class SyncService:
|
||||
f"error: {error_msg}. Falling back to manual count. "
|
||||
f"This will slow down watermark detection!"
|
||||
)
|
||||
# Track optimization failures for visibility
|
||||
logfire.metric_counter("sync.scan.file_count_failure").add(1)
|
||||
# Fallback: count using scan_directory
|
||||
count = 0
|
||||
async for _ in self.scan_directory(directory):
|
||||
@@ -1081,6 +1084,9 @@ class SyncService:
|
||||
This is dramatically faster than scanning all files and comparing mtimes,
|
||||
especially on network filesystems like TigrisFS where stat operations are expensive.
|
||||
|
||||
On Windows, subprocess is not supported with SelectorEventLoop (which we use
|
||||
to avoid aiosqlite cleanup issues), so we implement mtime filtering in Python.
|
||||
|
||||
Args:
|
||||
directory: Directory to scan
|
||||
since_timestamp: Unix timestamp to find files newer than
|
||||
@@ -1088,6 +1094,16 @@ class SyncService:
|
||||
Returns:
|
||||
List of relative file paths modified since the timestamp (respects .bmignore)
|
||||
"""
|
||||
# Windows with SelectorEventLoop doesn't support subprocess
|
||||
# Implement mtime filtering in Python to preserve watermark optimization
|
||||
if sys.platform == "win32":
|
||||
file_paths = []
|
||||
async for file_path_str, stat_info in self.scan_directory(directory):
|
||||
if stat_info.st_mtime > since_timestamp:
|
||||
rel_path = Path(file_path_str).relative_to(directory).as_posix()
|
||||
file_paths.append(rel_path)
|
||||
return file_paths
|
||||
|
||||
# Convert timestamp to find-compatible format
|
||||
since_date = datetime.fromtimestamp(since_timestamp).strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
@@ -1105,8 +1121,6 @@ class SyncService:
|
||||
f"error: {error_msg}. Falling back to full scan. "
|
||||
f"This will cause slow syncs on large projects!"
|
||||
)
|
||||
# Track optimization failures for visibility
|
||||
logfire.metric_counter("sync.scan.optimization_failure").add(1)
|
||||
# Fallback to full scan
|
||||
return await self._scan_directory_full(directory)
|
||||
|
||||
@@ -1206,8 +1220,8 @@ async def get_sync_service(project: Project) -> SyncService: # pragma: no cover
|
||||
|
||||
project_path = Path(project.path)
|
||||
entity_parser = EntityParser(project_path)
|
||||
markdown_processor = MarkdownProcessor(entity_parser)
|
||||
file_service = FileService(project_path, markdown_processor)
|
||||
markdown_processor = MarkdownProcessor(entity_parser, app_config=app_config)
|
||||
file_service = FileService(project_path, markdown_processor, app_config=app_config)
|
||||
|
||||
# Initialize repositories
|
||||
entity_repository = EntityRepository(session_maker, project_id=project.id)
|
||||
|
||||
@@ -5,7 +5,10 @@ import os
|
||||
from collections import defaultdict
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Set, Sequence
|
||||
from typing import List, Optional, Set, Sequence, Callable, Awaitable, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from basic_memory.sync.sync_service import SyncService
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, WATCH_STATUS_JSON
|
||||
from basic_memory.ignore_utils import load_gitignore_patterns, should_ignore_path
|
||||
@@ -71,12 +74,17 @@ class WatchServiceState(BaseModel):
|
||||
self.last_error = datetime.now()
|
||||
|
||||
|
||||
# Type alias for sync service factory function
|
||||
SyncServiceFactory = Callable[[Project], Awaitable["SyncService"]]
|
||||
|
||||
|
||||
class WatchService:
|
||||
def __init__(
|
||||
self,
|
||||
app_config: BasicMemoryConfig,
|
||||
project_repository: ProjectRepository,
|
||||
quiet: bool = False,
|
||||
sync_service_factory: Optional[SyncServiceFactory] = None,
|
||||
):
|
||||
self.app_config = app_config
|
||||
self.project_repository = project_repository
|
||||
@@ -84,10 +92,20 @@ class WatchService:
|
||||
self.status_path = Path.home() / ".basic-memory" / WATCH_STATUS_JSON
|
||||
self.status_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self._ignore_patterns_cache: dict[Path, Set[str]] = {}
|
||||
self._sync_service_factory = sync_service_factory
|
||||
|
||||
# quiet mode for mcp so it doesn't mess up stdout
|
||||
self.console = Console(quiet=quiet)
|
||||
|
||||
async def _get_sync_service(self, project: Project) -> "SyncService":
|
||||
"""Get sync service for a project, using factory if provided."""
|
||||
if self._sync_service_factory:
|
||||
return await self._sync_service_factory(project)
|
||||
# Fall back to default factory
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
return await get_sync_service(project)
|
||||
|
||||
async def _schedule_restart(self, stop_event: asyncio.Event):
|
||||
"""Schedule a restart of the watch service after the configured interval."""
|
||||
await asyncio.sleep(self.app_config.watch_project_reload_interval)
|
||||
@@ -233,9 +251,6 @@ class WatchService:
|
||||
|
||||
async def handle_changes(self, project: Project, changes: Set[FileChange]) -> None:
|
||||
"""Process a batch of file changes"""
|
||||
# avoid circular imports
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
# Check if project still exists in configuration before processing
|
||||
# This prevents deleted projects from being recreated by background sync
|
||||
from basic_memory.config import ConfigManager
|
||||
@@ -250,7 +265,7 @@ class WatchService:
|
||||
)
|
||||
return
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
sync_service = await self._get_sync_service(project)
|
||||
file_service = sync_service.file_service
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
@@ -0,0 +1,249 @@
|
||||
"""Anonymous telemetry for Basic Memory (Homebrew-style opt-out).
|
||||
|
||||
This module implements privacy-respecting usage analytics following the Homebrew model:
|
||||
- Telemetry is ON by default
|
||||
- Users can easily opt out: `bm telemetry disable`
|
||||
- First run shows a one-time notice (not a prompt)
|
||||
- Only anonymous data is collected (random UUID, no personal info)
|
||||
|
||||
What we collect:
|
||||
- App version, Python version, OS, architecture
|
||||
- Feature usage (which MCP tools and CLI commands are used)
|
||||
- Error types (sanitized, no file paths or personal data)
|
||||
|
||||
What we NEVER collect:
|
||||
- Note content, file names, or paths
|
||||
- Personal information
|
||||
- IP addresses (OpenPanel doesn't store these)
|
||||
|
||||
Documentation: https://basicmemory.com/telemetry
|
||||
"""
|
||||
|
||||
import platform
|
||||
import re
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from loguru import logger
|
||||
from openpanel import OpenPanel
|
||||
|
||||
from basic_memory import __version__
|
||||
|
||||
# --- Configuration ---
|
||||
|
||||
# OpenPanel credentials (write-only, safe to embed in client code)
|
||||
# These can only send events to our dashboard, not read any data
|
||||
OPENPANEL_CLIENT_ID = "2e7b036d-c6e5-40aa-91eb-5c70a8ef21a3"
|
||||
OPENPANEL_CLIENT_SECRET = "sec_92f7f8328bd0368ff4c2"
|
||||
|
||||
TELEMETRY_DOCS_URL = "https://basicmemory.com/telemetry"
|
||||
|
||||
TELEMETRY_NOTICE = f"""
|
||||
Basic Memory collects anonymous usage statistics to help improve the software.
|
||||
This includes: version, OS, feature usage, and errors. No personal data or note content.
|
||||
|
||||
To opt out: bm telemetry disable
|
||||
Details: {TELEMETRY_DOCS_URL}
|
||||
"""
|
||||
|
||||
# --- Module State ---
|
||||
|
||||
_client: OpenPanel | None = None
|
||||
_initialized: bool = False
|
||||
|
||||
|
||||
# --- Installation ID ---
|
||||
|
||||
|
||||
def get_install_id() -> str:
|
||||
"""Get or create anonymous installation ID.
|
||||
|
||||
Creates a random UUID on first run and stores it locally.
|
||||
User can delete ~/.basic-memory/.install_id to reset.
|
||||
"""
|
||||
id_file = Path.home() / ".basic-memory" / ".install_id"
|
||||
|
||||
if id_file.exists():
|
||||
return id_file.read_text().strip()
|
||||
|
||||
install_id = str(uuid.uuid4())
|
||||
id_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
id_file.write_text(install_id)
|
||||
return install_id
|
||||
|
||||
|
||||
# --- Client Management ---
|
||||
|
||||
|
||||
def _get_client() -> OpenPanel:
|
||||
"""Get or create the OpenPanel client (singleton).
|
||||
|
||||
Lazily initializes the client with global properties.
|
||||
"""
|
||||
global _client, _initialized
|
||||
|
||||
if _client is None:
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
config = ConfigManager().config
|
||||
|
||||
# Trigger: first call to track an event
|
||||
# Why: lazy init avoids work if telemetry never used; disabled flag
|
||||
# tells OpenPanel to skip network calls when user opts out or during tests
|
||||
# Outcome: client ready to queue events (or silently discard if disabled)
|
||||
is_disabled = not config.telemetry_enabled or config.is_test_env
|
||||
_client = OpenPanel(
|
||||
client_id=OPENPANEL_CLIENT_ID,
|
||||
client_secret=OPENPANEL_CLIENT_SECRET,
|
||||
disabled=is_disabled,
|
||||
)
|
||||
|
||||
if config.telemetry_enabled and not config.is_test_env and not _initialized:
|
||||
# Set global properties that go with every event
|
||||
_client.set_global_properties(
|
||||
{
|
||||
"app_version": __version__,
|
||||
"python_version": platform.python_version(),
|
||||
"os": platform.system().lower(),
|
||||
"arch": platform.machine(),
|
||||
"install_id": get_install_id(),
|
||||
"source": "foss",
|
||||
}
|
||||
)
|
||||
_initialized = True
|
||||
|
||||
return _client
|
||||
|
||||
|
||||
def reset_client() -> None:
|
||||
"""Reset the telemetry client (for testing or after config changes)."""
|
||||
global _client, _initialized
|
||||
_client = None
|
||||
_initialized = False
|
||||
|
||||
|
||||
# --- Event Tracking ---
|
||||
|
||||
|
||||
def track(event: str, properties: dict[str, Any] | None = None) -> None:
|
||||
"""Track an event. Fire-and-forget, never raises.
|
||||
|
||||
Args:
|
||||
event: Event name (e.g., "app_started", "mcp_tool_called")
|
||||
properties: Optional event properties
|
||||
"""
|
||||
# Constraint: telemetry must never break the application
|
||||
# Even if OpenPanel API is down or config is corrupt, user's command must succeed
|
||||
try:
|
||||
_get_client().track(event, properties or {})
|
||||
except Exception as e:
|
||||
logger.opt(exception=False).debug(f"Telemetry failed: {e}")
|
||||
|
||||
|
||||
# --- First-Run Notice ---
|
||||
|
||||
|
||||
def show_notice_if_needed() -> None:
|
||||
"""Show one-time telemetry notice (Homebrew style).
|
||||
|
||||
Only shows if:
|
||||
- Telemetry is enabled
|
||||
- Notice hasn't been shown before
|
||||
|
||||
After showing, marks the notice as shown in config.
|
||||
"""
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
config_manager = ConfigManager()
|
||||
config = config_manager.config
|
||||
|
||||
if config.telemetry_enabled and not config.telemetry_notice_shown:
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
|
||||
# Print to stderr so it doesn't interfere with command output
|
||||
console = Console(stderr=True)
|
||||
console.print(
|
||||
Panel(
|
||||
TELEMETRY_NOTICE.strip(),
|
||||
title="[dim]Telemetry Notice[/dim]",
|
||||
border_style="dim",
|
||||
expand=False,
|
||||
)
|
||||
)
|
||||
|
||||
# Mark as shown so we don't show again
|
||||
config.telemetry_notice_shown = True
|
||||
config_manager.save_config(config)
|
||||
|
||||
|
||||
# --- Convenience Functions ---
|
||||
|
||||
|
||||
def track_app_started(mode: str) -> None:
|
||||
"""Track app startup.
|
||||
|
||||
Args:
|
||||
mode: "cli" or "mcp"
|
||||
"""
|
||||
track("app_started", {"mode": mode})
|
||||
|
||||
|
||||
def track_mcp_tool(tool_name: str) -> None:
|
||||
"""Track MCP tool usage.
|
||||
|
||||
Args:
|
||||
tool_name: Name of the tool (e.g., "write_note", "search_notes")
|
||||
"""
|
||||
track("mcp_tool_called", {"tool": tool_name})
|
||||
|
||||
|
||||
def track_cli_command(command: str) -> None:
|
||||
"""Track CLI command usage.
|
||||
|
||||
Args:
|
||||
command: Command name (e.g., "sync", "import claude")
|
||||
"""
|
||||
track("cli_command", {"command": command})
|
||||
|
||||
|
||||
def track_sync_completed(entity_count: int, duration_ms: int) -> None:
|
||||
"""Track sync completion.
|
||||
|
||||
Args:
|
||||
entity_count: Number of entities synced
|
||||
duration_ms: Duration in milliseconds
|
||||
"""
|
||||
track("sync_completed", {"entity_count": entity_count, "duration_ms": duration_ms})
|
||||
|
||||
|
||||
def track_import_completed(source: str, count: int) -> None:
|
||||
"""Track import completion.
|
||||
|
||||
Args:
|
||||
source: Import source (e.g., "claude", "chatgpt")
|
||||
count: Number of items imported
|
||||
"""
|
||||
track("import_completed", {"source": source, "count": count})
|
||||
|
||||
|
||||
def track_error(error_type: str, message: str) -> None:
|
||||
"""Track an error (sanitized).
|
||||
|
||||
Args:
|
||||
error_type: Exception class name
|
||||
message: Error message (will be sanitized to remove file paths)
|
||||
"""
|
||||
if not message:
|
||||
track("error", {"type": error_type, "message": ""})
|
||||
return
|
||||
|
||||
# Sanitize file paths to prevent leaking user directory structure
|
||||
# Unix paths: /Users/name/file.py, /home/user/notes/doc.md
|
||||
sanitized = re.sub(r"/[\w/.+-]+\.\w+", "[FILE]", message)
|
||||
# Windows paths: C:\Users\name\file.py, D:\projects\doc.md
|
||||
sanitized = re.sub(r"[A-Z]:\\[\w\\.+-]+\.\w+", "[FILE]", sanitized, flags=re.IGNORECASE)
|
||||
|
||||
# Truncate to avoid sending too much data
|
||||
track("error", {"type": error_type, "message": sanitized[:200]})
|
||||
+67
-64
@@ -5,9 +5,9 @@ import os
|
||||
import logging
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Optional, Protocol, Union, runtime_checkable, List
|
||||
from typing import Protocol, Union, runtime_checkable, List
|
||||
|
||||
from loguru import logger
|
||||
from unidecode import unidecode
|
||||
@@ -67,9 +67,6 @@ class PathLike(Protocol):
|
||||
# This preserves compatibility with existing code while we migrate
|
||||
FilePath = Union[Path, str]
|
||||
|
||||
# Disable the "Queue is full" warning
|
||||
logging.getLogger("opentelemetry.sdk.metrics._internal.instrument").setLevel(logging.ERROR)
|
||||
|
||||
|
||||
def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: bool = True) -> str:
|
||||
"""Generate a stable permalink from a file path.
|
||||
@@ -206,29 +203,37 @@ def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: b
|
||||
|
||||
|
||||
def setup_logging(
|
||||
env: str,
|
||||
home_dir: Path,
|
||||
log_file: Optional[str] = None,
|
||||
log_level: str = "INFO",
|
||||
console: bool = True,
|
||||
log_to_file: bool = False,
|
||||
log_to_stdout: bool = False,
|
||||
structured_context: bool = False,
|
||||
) -> None: # pragma: no cover
|
||||
"""
|
||||
Configure logging for the application.
|
||||
"""Configure logging with explicit settings.
|
||||
|
||||
This function provides a simple, explicit interface for configuring logging.
|
||||
Each entry point (CLI, MCP, API) should call this with appropriate settings.
|
||||
|
||||
Args:
|
||||
env: The environment name (dev, test, prod)
|
||||
home_dir: The root directory for the application
|
||||
log_file: The name of the log file to write to
|
||||
log_level: The logging level to use
|
||||
console: Whether to log to the console
|
||||
log_level: DEBUG, INFO, WARNING, ERROR
|
||||
log_to_file: Write to ~/.basic-memory/basic-memory.log with rotation
|
||||
log_to_stdout: Write to stderr (for Docker/cloud deployments)
|
||||
structured_context: Bind tenant_id, fly_region, etc. for cloud observability
|
||||
"""
|
||||
# Remove default handler and any existing handlers
|
||||
logger.remove()
|
||||
|
||||
# Add file handler if we are not running tests and a log file is specified
|
||||
if log_file and env != "test":
|
||||
# Setup file logger
|
||||
log_path = home_dir / log_file
|
||||
# In test mode, only log to stdout regardless of settings
|
||||
env = os.getenv("BASIC_MEMORY_ENV", "dev")
|
||||
if env == "test":
|
||||
logger.add(sys.stderr, level=log_level, backtrace=True, diagnose=True, colorize=True)
|
||||
return
|
||||
|
||||
# Add file handler with rotation
|
||||
if log_to_file:
|
||||
log_path = Path.home() / ".basic-memory" / "basic-memory.log"
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
# Keep logging synchronous (enqueue=False) to avoid background logging threads.
|
||||
# Background threads are a common source of "hang on exit" issues in CLI/test runs.
|
||||
logger.add(
|
||||
str(log_path),
|
||||
level=log_level,
|
||||
@@ -236,42 +241,28 @@ def setup_logging(
|
||||
retention="10 days",
|
||||
backtrace=True,
|
||||
diagnose=True,
|
||||
enqueue=True,
|
||||
enqueue=False,
|
||||
colorize=False,
|
||||
)
|
||||
|
||||
# Add console logger if requested or in test mode
|
||||
if env == "test" or console:
|
||||
# Add stdout handler (for Docker/cloud)
|
||||
if log_to_stdout:
|
||||
logger.add(sys.stderr, level=log_level, backtrace=True, diagnose=True, colorize=True)
|
||||
|
||||
logger.info(f"ENV: '{env}' Log level: '{log_level}' Logging to {log_file}")
|
||||
|
||||
# Bind environment context for structured logging (works in both local and cloud)
|
||||
tenant_id = os.getenv("BASIC_MEMORY_TENANT_ID", "local")
|
||||
fly_app_name = os.getenv("FLY_APP_NAME", "local")
|
||||
fly_machine_id = os.getenv("FLY_MACHINE_ID", "local")
|
||||
fly_region = os.getenv("FLY_REGION", "local")
|
||||
|
||||
logger.configure(
|
||||
extra={
|
||||
"tenant_id": tenant_id,
|
||||
"fly_app_name": fly_app_name,
|
||||
"fly_machine_id": fly_machine_id,
|
||||
"fly_region": fly_region,
|
||||
}
|
||||
)
|
||||
# Bind structured context for cloud observability
|
||||
if structured_context:
|
||||
logger.configure(
|
||||
extra={
|
||||
"tenant_id": os.getenv("BASIC_MEMORY_TENANT_ID", "local"),
|
||||
"fly_app_name": os.getenv("FLY_APP_NAME", "local"),
|
||||
"fly_machine_id": os.getenv("FLY_MACHINE_ID", "local"),
|
||||
"fly_region": os.getenv("FLY_REGION", "local"),
|
||||
}
|
||||
)
|
||||
|
||||
# Reduce noise from third-party libraries
|
||||
noisy_loggers = {
|
||||
# HTTP client logs
|
||||
"httpx": logging.WARNING,
|
||||
# File watching logs
|
||||
"watchfiles.main": logging.WARNING,
|
||||
}
|
||||
|
||||
# Set log levels for noisy loggers
|
||||
for logger_name, level in noisy_loggers.items():
|
||||
logging.getLogger(logger_name).setLevel(level)
|
||||
logging.getLogger("httpx").setLevel(logging.WARNING)
|
||||
logging.getLogger("watchfiles.main").setLevel(logging.WARNING)
|
||||
|
||||
|
||||
def parse_tags(tags: Union[List[str], str, None]) -> List[str]:
|
||||
@@ -340,7 +331,7 @@ def normalize_file_path_for_comparison(file_path: str) -> str:
|
||||
This function normalizes file paths to help detect potential conflicts:
|
||||
- Converts to lowercase for case-insensitive comparison
|
||||
- Normalizes Unicode characters
|
||||
- Handles path separators consistently
|
||||
- Converts backslashes to forward slashes for cross-platform consistency
|
||||
|
||||
Args:
|
||||
file_path: The file path to normalize
|
||||
@@ -349,19 +340,15 @@ def normalize_file_path_for_comparison(file_path: str) -> str:
|
||||
Normalized file path for comparison purposes
|
||||
"""
|
||||
import unicodedata
|
||||
from pathlib import PureWindowsPath
|
||||
|
||||
# Convert to lowercase for case-insensitive comparison
|
||||
normalized = file_path.lower()
|
||||
# Use PureWindowsPath to ensure backslashes are treated as separators
|
||||
# regardless of current platform, then convert to POSIX-style
|
||||
normalized = PureWindowsPath(file_path).as_posix().lower()
|
||||
|
||||
# Normalize Unicode characters (NFD normalization)
|
||||
normalized = unicodedata.normalize("NFD", normalized)
|
||||
|
||||
# Replace path separators with forward slashes
|
||||
normalized = normalized.replace("\\", "/")
|
||||
|
||||
# Remove multiple slashes
|
||||
normalized = re.sub(r"/+", "/", normalized)
|
||||
|
||||
return normalized
|
||||
|
||||
|
||||
@@ -445,21 +432,37 @@ def validate_project_path(path: str, project_path: Path) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def ensure_timezone_aware(dt: datetime) -> datetime:
|
||||
"""Ensure a datetime is timezone-aware using system timezone.
|
||||
def ensure_timezone_aware(dt: datetime, cloud_mode: bool | None = None) -> datetime:
|
||||
"""Ensure a datetime is timezone-aware.
|
||||
|
||||
If the datetime is naive, convert it to timezone-aware using the system's local timezone.
|
||||
If it's already timezone-aware, return it unchanged.
|
||||
If the datetime is naive, convert it to timezone-aware. The interpretation
|
||||
depends on cloud_mode:
|
||||
- In cloud mode (PostgreSQL/asyncpg): naive datetimes are interpreted as UTC
|
||||
- In local mode (SQLite): naive datetimes are interpreted as local time
|
||||
|
||||
asyncpg uses binary protocol which returns timestamps in UTC but as naive
|
||||
datetimes. In cloud deployments, cloud_mode=True handles this correctly.
|
||||
|
||||
Args:
|
||||
dt: The datetime to ensure is timezone-aware
|
||||
cloud_mode: Optional explicit cloud_mode setting. If None, loads from config.
|
||||
|
||||
Returns:
|
||||
A timezone-aware datetime
|
||||
"""
|
||||
if dt.tzinfo is None:
|
||||
# Naive datetime - assume it's in local time and add timezone
|
||||
return dt.astimezone()
|
||||
# Determine cloud_mode: use explicit parameter if provided, otherwise load from config
|
||||
if cloud_mode is None:
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
cloud_mode = ConfigManager().config.cloud_mode_enabled
|
||||
|
||||
if cloud_mode:
|
||||
# Cloud/PostgreSQL mode: naive datetimes from asyncpg are already UTC
|
||||
return dt.replace(tzinfo=timezone.utc)
|
||||
else:
|
||||
# Local/SQLite mode: naive datetimes are in local time
|
||||
return dt.astimezone()
|
||||
else:
|
||||
# Already timezone-aware
|
||||
return dt
|
||||
|
||||
+86
-48
@@ -50,18 +50,23 @@ The `app` fixture ensures FastAPI dependency overrides are active, and
|
||||
`mcp_server` provides the MCP server with proper project session initialization.
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import AsyncGenerator, Literal
|
||||
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from pathlib import Path
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||
from sqlalchemy.pool import NullPool
|
||||
from testcontainers.postgres import PostgresContainer
|
||||
|
||||
from httpx import AsyncClient, ASGITransport
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, ProjectConfig, ConfigManager, DatabaseBackend
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.models.base import Base
|
||||
from basic_memory.repository.project_repository import ProjectRepository
|
||||
from fastapi import FastAPI
|
||||
|
||||
@@ -72,25 +77,38 @@ from basic_memory.deps import get_project_config, get_engine_factory, get_app_co
|
||||
from basic_memory.mcp import tools # noqa: F401
|
||||
|
||||
|
||||
@pytest.fixture(
|
||||
params=[
|
||||
pytest.param("sqlite", id="sqlite"),
|
||||
pytest.param("postgres", id="postgres", marks=pytest.mark.postgres),
|
||||
]
|
||||
)
|
||||
def db_backend(request) -> Literal["sqlite", "postgres"]:
|
||||
"""Parametrize tests to run against both SQLite and Postgres.
|
||||
# =============================================================================
|
||||
# Database Backend Selection (env var approach)
|
||||
# =============================================================================
|
||||
# By default, integration tests run against SQLite.
|
||||
# Set BASIC_MEMORY_TEST_POSTGRES=1 to run against Postgres (uses testcontainers).
|
||||
|
||||
Usage:
|
||||
pytest # Runs tests against SQLite only (default)
|
||||
pytest -m postgres # Runs tests against Postgres only
|
||||
pytest -m "not postgres" # Runs tests against SQLite only
|
||||
pytest --run-all-backends # Runs tests against both backends
|
||||
|
||||
Note: Only tests that use database fixtures (engine_factory, session_maker, etc.)
|
||||
will be parametrized. Tests that don't use the database won't be affected.
|
||||
@pytest.fixture(scope="session")
|
||||
def db_backend() -> Literal["sqlite", "postgres"]:
|
||||
"""Determine database backend from environment variable.
|
||||
|
||||
Default: sqlite
|
||||
Set BASIC_MEMORY_TEST_POSTGRES=1 to use postgres
|
||||
"""
|
||||
return request.param
|
||||
if os.environ.get("BASIC_MEMORY_TEST_POSTGRES", "").lower() in ("1", "true", "yes"):
|
||||
return "postgres"
|
||||
return "sqlite"
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def postgres_container(db_backend):
|
||||
"""Session-scoped Postgres container for integration tests.
|
||||
|
||||
Uses testcontainers to spin up a real Postgres instance.
|
||||
Only starts if db_backend is "postgres".
|
||||
"""
|
||||
if db_backend != "postgres":
|
||||
yield None
|
||||
return
|
||||
|
||||
with PostgresContainer("postgres:16-alpine") as postgres:
|
||||
yield postgres
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
@@ -98,50 +116,65 @@ async def engine_factory(
|
||||
app_config,
|
||||
config_manager,
|
||||
db_backend: Literal["sqlite", "postgres"],
|
||||
postgres_container,
|
||||
tmp_path,
|
||||
) -> AsyncGenerator[tuple, None]:
|
||||
"""Create engine and session factory for the configured database backend."""
|
||||
from basic_memory.models.search import CREATE_SEARCH_INDEX
|
||||
from basic_memory.models.search import (
|
||||
CREATE_SEARCH_INDEX,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_TABLE,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_FTS,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_METADATA,
|
||||
)
|
||||
from basic_memory import db
|
||||
|
||||
# Determine database type based on backend
|
||||
if db_backend == "postgres":
|
||||
db_type = DatabaseType.FILESYSTEM
|
||||
else:
|
||||
db_type = DatabaseType.FILESYSTEM # Integration tests use file-based SQLite
|
||||
# Postgres mode using testcontainers
|
||||
sync_url = postgres_container.get_connection_url()
|
||||
async_url = sync_url.replace("postgresql+psycopg2", "postgresql+asyncpg")
|
||||
|
||||
# Use tmp_path for SQLite, use config database_path for Postgres
|
||||
if db_backend == "sqlite":
|
||||
db_path = tmp_path / "test.db"
|
||||
else:
|
||||
db_path = app_config.database_path
|
||||
engine = create_async_engine(
|
||||
async_url,
|
||||
echo=False,
|
||||
poolclass=NullPool,
|
||||
)
|
||||
|
||||
if db_backend == "postgres":
|
||||
# Postgres: Create fresh engine for each test with full schema reset
|
||||
config_manager._config = app_config
|
||||
session_maker = async_sessionmaker(
|
||||
bind=engine,
|
||||
class_=AsyncSession,
|
||||
expire_on_commit=False,
|
||||
autoflush=False,
|
||||
)
|
||||
|
||||
# Use context manager to handle engine disposal properly
|
||||
async with engine_session_factory(db_path, db_type) as (engine, session_maker):
|
||||
# Drop and recreate schema for complete isolation
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(text("DROP SCHEMA IF EXISTS public CASCADE"))
|
||||
await conn.execute(text("CREATE SCHEMA public"))
|
||||
await conn.execute(text("GRANT ALL ON SCHEMA public TO basic_memory_user"))
|
||||
await conn.execute(text("GRANT ALL ON SCHEMA public TO public"))
|
||||
# Set module-level state to prevent MCP lifespan from re-initializing
|
||||
# This ensures get_or_create_db() sees an existing engine and skips initialization
|
||||
db._engine = engine
|
||||
db._session_maker = session_maker
|
||||
|
||||
# Run migrations to create production tables
|
||||
from basic_memory.db import run_migrations
|
||||
# Drop and recreate all tables for test isolation
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(text("DROP TABLE IF EXISTS search_index CASCADE"))
|
||||
await conn.run_sync(Base.metadata.drop_all)
|
||||
await conn.run_sync(Base.metadata.create_all)
|
||||
# asyncpg requires separate execute calls for each statement
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_TABLE)
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_FTS)
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_METADATA)
|
||||
|
||||
await run_migrations(app_config, db_type)
|
||||
yield engine, session_maker
|
||||
|
||||
yield engine, session_maker
|
||||
# Clean up module-level state
|
||||
await engine.dispose()
|
||||
db._engine = None
|
||||
db._session_maker = None
|
||||
|
||||
else:
|
||||
# SQLite: Create fresh database (fast with tmp files)
|
||||
db_path = tmp_path / "test.db"
|
||||
db_type = DatabaseType.FILESYSTEM
|
||||
|
||||
async with engine_session_factory(db_path, db_type) as (engine, session_maker):
|
||||
# Create all tables via ORM
|
||||
from basic_memory.models.base import Base
|
||||
|
||||
async with engine.begin() as conn:
|
||||
await conn.run_sync(Base.metadata.create_all)
|
||||
|
||||
@@ -181,7 +214,11 @@ def config_home(tmp_path, monkeypatch) -> Path:
|
||||
|
||||
@pytest.fixture
|
||||
def app_config(
|
||||
config_home, db_backend: Literal["sqlite", "postgres"], tmp_path, monkeypatch
|
||||
config_home,
|
||||
db_backend: Literal["sqlite", "postgres"],
|
||||
postgres_container,
|
||||
tmp_path,
|
||||
monkeypatch,
|
||||
) -> BasicMemoryConfig:
|
||||
"""Create test app configuration."""
|
||||
# Disable cloud mode for CLI tests
|
||||
@@ -190,12 +227,12 @@ def app_config(
|
||||
# Create a basic config with test-project like unit tests do
|
||||
projects = {"test-project": str(config_home)}
|
||||
|
||||
# Configure database backend based on test parameter
|
||||
# Configure database backend based on env var
|
||||
if db_backend == "postgres":
|
||||
database_backend = DatabaseBackend.POSTGRES
|
||||
database_url = (
|
||||
"postgresql+asyncpg://basic_memory_user:dev_password@localhost:5433/basic_memory_test"
|
||||
)
|
||||
# Get URL from testcontainer and convert to asyncpg driver
|
||||
sync_url = postgres_container.get_connection_url()
|
||||
database_url = sync_url.replace("postgresql+psycopg2", "postgresql+asyncpg")
|
||||
else:
|
||||
database_backend = DatabaseBackend.SQLITE
|
||||
database_url = None
|
||||
@@ -207,6 +244,7 @@ def app_config(
|
||||
default_project_mode=False, # Match real-world usage - tools must pass explicit project
|
||||
update_permalinks_on_move=True,
|
||||
cloud_mode=False, # Explicitly disable cloud mode
|
||||
sync_changes=False, # Disable file sync in tests - prevents lifespan from starting blocking task
|
||||
database_backend=database_backend,
|
||||
database_url=database_url,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
"""
|
||||
Integration test for FastAPI lifespan shutdown behavior.
|
||||
|
||||
This test verifies the asyncio cancellation pattern used by the API lifespan:
|
||||
when the background sync task is cancelled during shutdown, it must be *awaited*
|
||||
before database shutdown begins. This prevents "hang on exit" scenarios in
|
||||
`asyncio.run(...)` callers (e.g. CLI/MCP clients using httpx ASGITransport).
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
|
||||
from httpx import ASGITransport, AsyncClient
|
||||
|
||||
|
||||
def test_lifespan_shutdown_awaits_sync_task_cancellation(app, monkeypatch):
|
||||
"""
|
||||
Ensure lifespan shutdown awaits the cancelled background sync task.
|
||||
|
||||
Why this is deterministic:
|
||||
- Cancelling a task does not make it "done" immediately; it becomes done only
|
||||
once the event loop schedules it and it processes the CancelledError.
|
||||
- In the buggy version, shutdown proceeded directly to db.shutdown_db()
|
||||
immediately after calling cancel(), so at *entry* to shutdown_db the task
|
||||
is still not done.
|
||||
- In the fixed version, lifespan does `await sync_task` before shutdown_db,
|
||||
so by the time shutdown_db is called, the task is done (cancelled).
|
||||
"""
|
||||
|
||||
# Import the *module* (not the package-level FastAPI `basic_memory.api.app` export)
|
||||
# so monkeypatching affects the exact symbols referenced inside lifespan().
|
||||
#
|
||||
# Note: `basic_memory/api/__init__.py` re-exports `app`, so `import basic_memory.api.app`
|
||||
# can resolve to the FastAPI instance rather than the `basic_memory.api.app` module.
|
||||
import importlib
|
||||
|
||||
api_app_module = importlib.import_module("basic_memory.api.app")
|
||||
|
||||
# Keep startup cheap: we don't need real DB init for this ordering test.
|
||||
async def _noop_initialize_app(_app_config):
|
||||
return None
|
||||
|
||||
async def _fake_get_or_create_db(*_args, **_kwargs):
|
||||
return object(), object()
|
||||
|
||||
monkeypatch.setattr(api_app_module, "initialize_app", _noop_initialize_app)
|
||||
monkeypatch.setattr(api_app_module.db, "get_or_create_db", _fake_get_or_create_db)
|
||||
|
||||
# Make the sync task long-lived so it must be cancelled on shutdown.
|
||||
async def _fake_initialize_file_sync(_app_config):
|
||||
await asyncio.Event().wait()
|
||||
|
||||
monkeypatch.setattr(api_app_module, "initialize_file_sync", _fake_initialize_file_sync)
|
||||
|
||||
# Assert ordering: shutdown_db must be called only after the sync_task is done.
|
||||
async def _assert_sync_task_done_before_db_shutdown():
|
||||
assert api_app_module.app.state.sync_task is not None
|
||||
assert api_app_module.app.state.sync_task.done()
|
||||
|
||||
monkeypatch.setattr(api_app_module.db, "shutdown_db", _assert_sync_task_done_before_db_shutdown)
|
||||
|
||||
async def _run_client_once():
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://test") as client:
|
||||
# Any request is sufficient to trigger lifespan startup/shutdown.
|
||||
await client.get("/__nonexistent__")
|
||||
|
||||
# Use asyncio.run to match the CLI/MCP execution model where loop teardown
|
||||
# would hang if a background task is left running.
|
||||
asyncio.run(_run_client_once())
|
||||
@@ -56,7 +56,7 @@ async def test_project_metadata_consistency(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_project_basic_operation(mcp_server, app, test_project):
|
||||
async def test_create_project_basic_operation(mcp_server, app, test_project, tmp_path):
|
||||
"""Test creating a new project with basic parameters."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -65,7 +65,9 @@ async def test_create_project_basic_operation(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "test-new-project",
|
||||
"project_path": "/tmp/test-new-project",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-test-new-project"
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -88,7 +90,7 @@ async def test_create_project_basic_operation(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_project_with_default_flag(mcp_server, app, test_project):
|
||||
async def test_create_project_with_default_flag(mcp_server, app, test_project, tmp_path):
|
||||
"""Test creating a project and setting it as default."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -97,7 +99,9 @@ async def test_create_project_with_default_flag(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "test-default-project",
|
||||
"project_path": "/tmp/test-default-project",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-test-default-project"
|
||||
),
|
||||
"set_default": True,
|
||||
},
|
||||
)
|
||||
@@ -116,7 +120,7 @@ async def test_create_project_with_default_flag(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_project_duplicate_name(mcp_server, app, test_project):
|
||||
async def test_create_project_duplicate_name(mcp_server, app, test_project, tmp_path):
|
||||
"""Test creating a project with duplicate name shows error."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -125,7 +129,9 @@ async def test_create_project_duplicate_name(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "duplicate-test",
|
||||
"project_path": "/tmp/duplicate-test-1",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-duplicate-test-1"
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -135,7 +141,9 @@ async def test_create_project_duplicate_name(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "duplicate-test",
|
||||
"project_path": "/tmp/duplicate-test-2",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-duplicate-test-2"
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -150,7 +158,7 @@ async def test_create_project_duplicate_name(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_project_basic_operation(mcp_server, app, test_project):
|
||||
async def test_delete_project_basic_operation(mcp_server, app, test_project, tmp_path):
|
||||
"""Test deleting a project that exists."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -159,7 +167,9 @@ async def test_delete_project_basic_operation(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "to-be-deleted",
|
||||
"project_path": "/tmp/to-be-deleted",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-to-be-deleted"
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -240,12 +250,14 @@ async def test_delete_current_project_protection(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_lifecycle_workflow(mcp_server, app, test_project):
|
||||
async def test_project_lifecycle_workflow(mcp_server, app, test_project, tmp_path):
|
||||
"""Test complete project lifecycle: create, switch, use, delete."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
project_name = "lifecycle-test"
|
||||
project_path = "/tmp/lifecycle-test"
|
||||
project_path = str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-lifecycle-test"
|
||||
)
|
||||
|
||||
# 1. Create new project
|
||||
create_result = await client.call_tool(
|
||||
@@ -295,7 +307,7 @@ async def test_project_lifecycle_workflow(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_delete_project_edge_cases(mcp_server, app, test_project):
|
||||
async def test_create_delete_project_edge_cases(mcp_server, app, test_project, tmp_path):
|
||||
"""Test edge cases for create and delete project operations."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -307,7 +319,11 @@ async def test_create_delete_project_edge_cases(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": special_name,
|
||||
"project_path": "/tmp/test-project-with-special-chars",
|
||||
"project_path": str(
|
||||
tmp_path.parent
|
||||
/ (tmp_path.name + "-projects")
|
||||
/ "project-test-project-with-special-chars"
|
||||
),
|
||||
},
|
||||
)
|
||||
assert "✓" in create_result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
@@ -333,7 +349,7 @@ async def test_create_delete_project_edge_cases(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_case_insensitive_project_switching(mcp_server, app, test_project):
|
||||
async def test_case_insensitive_project_switching(mcp_server, app, test_project, tmp_path):
|
||||
"""Test case-insensitive project switching with proper database lookup."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -343,7 +359,9 @@ async def test_case_insensitive_project_switching(mcp_server, app, test_project)
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": project_name,
|
||||
"project_path": f"/tmp/{project_name}",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / f"project-{project_name}"
|
||||
),
|
||||
},
|
||||
)
|
||||
assert "✓" in create_result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
@@ -384,7 +402,7 @@ async def test_case_insensitive_project_switching(mcp_server, app, test_project)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_case_insensitive_project_operations(mcp_server, app, test_project):
|
||||
async def test_case_insensitive_project_operations(mcp_server, app, test_project, tmp_path):
|
||||
"""Test that all project operations work correctly after case-insensitive switching."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -394,7 +412,9 @@ async def test_case_insensitive_project_operations(mcp_server, app, test_project
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": project_name,
|
||||
"project_path": f"/tmp/{project_name}",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / f"project-{project_name}"
|
||||
),
|
||||
},
|
||||
)
|
||||
assert "✓" in create_result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
@@ -465,7 +485,7 @@ async def test_case_insensitive_error_handling(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_case_preservation_in_project_list(mcp_server, app, test_project):
|
||||
async def test_case_preservation_in_project_list(mcp_server, app, test_project, tmp_path):
|
||||
"""Test that project names preserve their original case in listings."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
@@ -483,7 +503,9 @@ async def test_case_preservation_in_project_list(mcp_server, app, test_project):
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": project_name,
|
||||
"project_path": f"/tmp/{project_name}",
|
||||
"project_path": str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / f"project-{project_name}"
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -516,13 +538,15 @@ async def test_case_preservation_in_project_list(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_nested_project_paths_rejected(mcp_server, app, test_project):
|
||||
async def test_nested_project_paths_rejected(mcp_server, app, test_project, tmp_path):
|
||||
"""Test that creating nested project paths is rejected with clear error message."""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
# Create a parent project
|
||||
parent_name = "parent-project"
|
||||
parent_path = "/tmp/nested-test/parent"
|
||||
parent_path = str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-nested-test/parent"
|
||||
)
|
||||
|
||||
await client.call_tool(
|
||||
"create_memory_project",
|
||||
@@ -534,7 +558,9 @@ async def test_nested_project_paths_rejected(mcp_server, app, test_project):
|
||||
|
||||
# Try to create a child project nested under the parent
|
||||
child_name = "child-project"
|
||||
child_path = "/tmp/nested-test/parent/child"
|
||||
child_path = str(
|
||||
tmp_path.parent / (tmp_path.name + "-projects") / "project-nested-test/parent/child"
|
||||
)
|
||||
|
||||
with pytest.raises(Exception) as exc_info:
|
||||
await client.call_tool(
|
||||
|
||||
@@ -16,7 +16,7 @@ from fastmcp import Client
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_state_sync_after_default_change(
|
||||
mcp_server, app, config_manager, test_project
|
||||
mcp_server, app, config_manager, test_project, tmp_path
|
||||
):
|
||||
"""Test that MCP session stays in sync when default project is changed."""
|
||||
|
||||
@@ -26,7 +26,7 @@ async def test_project_state_sync_after_default_change(
|
||||
"create_memory_project",
|
||||
{
|
||||
"project_name": "minerva",
|
||||
"project_path": "/tmp/minerva-test-project",
|
||||
"project_path": str(tmp_path.parent / (tmp_path.name + "-projects") / "minerva"),
|
||||
"set_default": False, # Don't set as default yet
|
||||
},
|
||||
)
|
||||
|
||||
@@ -46,3 +46,57 @@ async def test_read_note_after_write(mcp_server, app, test_project):
|
||||
assert "# Test Note" in result_text
|
||||
assert "This is test content." in result_text
|
||||
assert "test/test-note" in result_text # permalink
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_read_note_underscored_folder_by_permalink(mcp_server, app, test_project):
|
||||
"""Test read_note with permalink from underscored folder.
|
||||
|
||||
Reproduces bug #416: read_note fails to find notes when given permalinks
|
||||
from underscored folder names (e.g., _archive/, _drafts/), even though
|
||||
the permalink is copied directly from the note's YAML frontmatter.
|
||||
"""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
# Create a note in an underscored folder
|
||||
write_result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": "Example Note",
|
||||
"folder": "_archive/articles",
|
||||
"content": "# Example Note\n\nThis is a test note in an underscored folder.",
|
||||
"tags": "test,archive",
|
||||
},
|
||||
)
|
||||
|
||||
assert len(write_result.content) == 1
|
||||
assert write_result.content[0].type == "text"
|
||||
write_text = write_result.content[0].text
|
||||
|
||||
# Verify the file path includes the underscore
|
||||
assert "_archive/articles/Example Note.md" in write_text
|
||||
|
||||
# Verify the permalink has underscores stripped (this is the expected behavior)
|
||||
assert "archive/articles/example-note" in write_text
|
||||
|
||||
# Now try to read the note using the permalink (without underscores)
|
||||
# This is the exact scenario from the bug report - using the permalink
|
||||
# that was generated in the YAML frontmatter
|
||||
read_result = await client.call_tool(
|
||||
"read_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"identifier": "archive/articles/example-note", # permalink without underscores
|
||||
},
|
||||
)
|
||||
|
||||
# This should succeed - the note should be found by its permalink
|
||||
assert len(read_result.content) == 1
|
||||
assert read_result.content[0].type == "text"
|
||||
result_text = read_result.content[0].text
|
||||
|
||||
# Should contain the note content
|
||||
assert "# Example Note" in result_text
|
||||
assert "This is a test note in an underscored folder." in result_text
|
||||
assert "archive/articles/example-note" in result_text # permalink
|
||||
|
||||
@@ -5,7 +5,6 @@ and other SQLite configuration settings work correctly in production scenarios.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import patch
|
||||
from sqlalchemy import text
|
||||
|
||||
|
||||
@@ -142,23 +141,6 @@ async def test_null_pool_on_windows(tmp_path, monkeypatch):
|
||||
assert isinstance(engine.pool, NullPool)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skipif(
|
||||
__import__("os").name == "nt", reason="Non-Windows test - cannot mock POSIX paths on Windows"
|
||||
)
|
||||
async def test_regular_pool_on_non_windows(tmp_path):
|
||||
"""Test that regular pooling is used on non-Windows platforms."""
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
db_path = tmp_path / "test_posix_pool.db"
|
||||
|
||||
with patch("basic_memory.db.os.name", "posix"):
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (engine, _):
|
||||
# Engine should NOT be using NullPool on non-Windows
|
||||
assert not isinstance(engine.pool, NullPool)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.windows
|
||||
@pytest.mark.skipif(
|
||||
|
||||
@@ -1,372 +0,0 @@
|
||||
"""
|
||||
Performance benchmark tests for sync operations.
|
||||
|
||||
These tests measure baseline performance for indexing operations to track
|
||||
improvements from optimizations. Tests are marked with @pytest.mark.benchmark
|
||||
and can be run separately.
|
||||
|
||||
Usage:
|
||||
# Run all benchmarks
|
||||
pytest test-int/test_sync_performance_benchmark.py -v
|
||||
|
||||
# Run specific benchmark
|
||||
pytest test-int/test_sync_performance_benchmark.py::test_benchmark_sync_100_files -v
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from pathlib import Path
|
||||
from textwrap import dedent
|
||||
|
||||
import pytest
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, ProjectConfig
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
|
||||
async def create_benchmark_file(path: Path, file_num: int, total_files: int) -> None:
|
||||
"""Create a realistic test markdown file with observations and relations.
|
||||
|
||||
Args:
|
||||
path: Path to create the file at
|
||||
file_num: Current file number (for unique content)
|
||||
total_files: Total number of files being created (for relation targets)
|
||||
"""
|
||||
# Create realistic content with varying complexity
|
||||
has_relations = file_num < (total_files - 1) # Most files have relations
|
||||
num_observations = min(3 + (file_num % 5), 10) # 3-10 observations per file
|
||||
|
||||
# Generate relation targets (some will be forward references)
|
||||
relations = []
|
||||
if has_relations:
|
||||
# Reference 1-3 other files
|
||||
num_relations = min(1 + (file_num % 3), 3)
|
||||
for i in range(num_relations):
|
||||
target_num = (file_num + i + 1) % total_files
|
||||
relations.append(f"- relates_to [[test-file-{target_num:04d}]]")
|
||||
|
||||
content = dedent(f"""
|
||||
---
|
||||
type: note
|
||||
tags: [benchmark, test, category-{file_num % 10}]
|
||||
---
|
||||
# Test File {file_num:04d}
|
||||
|
||||
This is benchmark test file {file_num} of {total_files}.
|
||||
It contains realistic markdown content to simulate actual usage.
|
||||
|
||||
## Observations
|
||||
{chr(10).join([f"- [category-{i % 5}] Observation {i} for file {file_num} with some content #tag{i}" for i in range(num_observations)])}
|
||||
|
||||
## Relations
|
||||
{chr(10).join(relations) if relations else "- No relations for this file"}
|
||||
|
||||
## Additional Content
|
||||
|
||||
This section contains additional prose to simulate real documents.
|
||||
Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed do eiusmod
|
||||
tempor incididunt ut labore et dolore magna aliqua.
|
||||
|
||||
### Subsection
|
||||
|
||||
More content here to make the file realistic. This helps test the
|
||||
full indexing pipeline including content extraction and search indexing.
|
||||
""").strip()
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
|
||||
|
||||
async def generate_benchmark_files(project_dir: Path, num_files: int) -> None:
|
||||
"""Generate benchmark test files.
|
||||
|
||||
Args:
|
||||
project_dir: Directory to create files in
|
||||
num_files: Number of files to generate
|
||||
"""
|
||||
print(f"\nGenerating {num_files} test files...")
|
||||
start = time.time()
|
||||
|
||||
# Create files in batches for faster generation
|
||||
batch_size = 100
|
||||
for batch_start in range(0, num_files, batch_size):
|
||||
batch_end = min(batch_start + batch_size, num_files)
|
||||
tasks = [
|
||||
create_benchmark_file(
|
||||
project_dir / f"category-{i % 10}" / f"test-file-{i:04d}.md", i, num_files
|
||||
)
|
||||
for i in range(batch_start, batch_end)
|
||||
]
|
||||
await asyncio.gather(*tasks)
|
||||
print(f" Created files {batch_start}-{batch_end} ({batch_end}/{num_files})")
|
||||
|
||||
duration = time.time() - start
|
||||
print(f" File generation completed in {duration:.2f}s ({num_files / duration:.1f} files/sec)")
|
||||
|
||||
|
||||
def get_db_size(db_path: Path) -> tuple[int, str]:
|
||||
"""Get database file size.
|
||||
|
||||
Returns:
|
||||
Tuple of (size_bytes, formatted_size)
|
||||
"""
|
||||
if not db_path.exists():
|
||||
return 0, "0 B"
|
||||
|
||||
size_bytes = db_path.stat().st_size
|
||||
|
||||
# Format size
|
||||
for unit in ["B", "KB", "MB", "GB"]:
|
||||
if size_bytes < 1024.0:
|
||||
return size_bytes, f"{size_bytes:.2f} {unit}"
|
||||
size_bytes /= 1024.0
|
||||
|
||||
return int(size_bytes * 1024**4), f"{size_bytes:.2f} TB"
|
||||
|
||||
|
||||
async def run_sync_benchmark(
|
||||
project_config: ProjectConfig, app_config: BasicMemoryConfig, num_files: int, test_name: str
|
||||
) -> dict:
|
||||
"""Run a sync benchmark and collect metrics.
|
||||
|
||||
Args:
|
||||
project_config: Project configuration
|
||||
app_config: App configuration
|
||||
num_files: Number of files to benchmark
|
||||
test_name: Name of the test for reporting
|
||||
|
||||
Returns:
|
||||
Dictionary with benchmark results
|
||||
"""
|
||||
project_dir = project_config.home
|
||||
db_path = app_config.database_path
|
||||
|
||||
print(f"\n{'=' * 70}")
|
||||
print(f"BENCHMARK: {test_name}")
|
||||
print(f"{'=' * 70}")
|
||||
|
||||
# Generate test files
|
||||
await generate_benchmark_files(project_dir, num_files)
|
||||
|
||||
# Get initial DB size
|
||||
initial_db_size, initial_db_formatted = get_db_size(db_path)
|
||||
print(f"\nInitial database size: {initial_db_formatted}")
|
||||
|
||||
# Create sync service
|
||||
from basic_memory.repository import ProjectRepository
|
||||
from basic_memory import db
|
||||
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
# Get or create project
|
||||
projects = await project_repository.find_all()
|
||||
if projects:
|
||||
project = projects[0]
|
||||
else:
|
||||
project = await project_repository.create(
|
||||
{
|
||||
"name": project_config.name,
|
||||
"path": str(project_config.home),
|
||||
"is_active": True,
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
|
||||
# Initialize search index (required for FTS5 table)
|
||||
await sync_service.search_service.init_search_index()
|
||||
|
||||
# Run sync and measure time
|
||||
print(f"\nStarting sync of {num_files} files...")
|
||||
sync_start = time.time()
|
||||
|
||||
report = await sync_service.sync(project_dir, project_name=project.name)
|
||||
|
||||
sync_duration = time.time() - sync_start
|
||||
|
||||
# Get final DB size
|
||||
final_db_size, final_db_formatted = get_db_size(db_path)
|
||||
db_growth = final_db_size - initial_db_size
|
||||
db_growth_formatted = f"{db_growth / 1024 / 1024:.2f} MB"
|
||||
|
||||
# Calculate metrics
|
||||
files_per_sec = num_files / sync_duration if sync_duration > 0 else 0
|
||||
ms_per_file = (sync_duration * 1000) / num_files if num_files > 0 else 0
|
||||
|
||||
# Print results
|
||||
print(f"\n{'-' * 70}")
|
||||
print("RESULTS:")
|
||||
print(f"{'-' * 70}")
|
||||
print(f"Files processed: {num_files}")
|
||||
print(f" New: {len(report.new)}")
|
||||
print(f" Modified: {len(report.modified)}")
|
||||
print(f" Deleted: {len(report.deleted)}")
|
||||
print(f" Moved: {len(report.moves)}")
|
||||
print("\nPerformance:")
|
||||
print(f" Total time: {sync_duration:.2f}s")
|
||||
print(f" Files/sec: {files_per_sec:.1f}")
|
||||
print(f" ms/file: {ms_per_file:.1f}")
|
||||
print("\nDatabase:")
|
||||
print(f" Initial size: {initial_db_formatted}")
|
||||
print(f" Final size: {final_db_formatted}")
|
||||
print(f" Growth: {db_growth_formatted}")
|
||||
print(f" Growth per file: {(db_growth / num_files / 1024):.2f} KB")
|
||||
print(f"{'=' * 70}\n")
|
||||
|
||||
return {
|
||||
"test_name": test_name,
|
||||
"num_files": num_files,
|
||||
"sync_duration_sec": sync_duration,
|
||||
"files_per_sec": files_per_sec,
|
||||
"ms_per_file": ms_per_file,
|
||||
"new_files": len(report.new),
|
||||
"modified_files": len(report.modified),
|
||||
"deleted_files": len(report.deleted),
|
||||
"moved_files": len(report.moves),
|
||||
"initial_db_size": initial_db_size,
|
||||
"final_db_size": final_db_size,
|
||||
"db_growth_bytes": db_growth,
|
||||
"db_growth_per_file_bytes": db_growth / num_files if num_files > 0 else 0,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
async def test_benchmark_sync_100_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 100 files (small repository)."""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=100, test_name="Sync 100 files (small repository)"
|
||||
)
|
||||
|
||||
# Basic assertions to ensure sync worked
|
||||
# Note: May be slightly more than 100 due to OS-generated files (.DS_Store, etc.)
|
||||
assert results["new_files"] >= 100
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skip
|
||||
async def test_benchmark_sync_500_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 500 files (medium repository)."""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=500, test_name="Sync 500 files (medium repository)"
|
||||
)
|
||||
|
||||
# Basic assertions
|
||||
# Note: May be slightly more than 500 due to OS-generated files
|
||||
assert results["new_files"] >= 500
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.slow
|
||||
@pytest.mark.skip
|
||||
async def test_benchmark_sync_1000_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 1000 files (large repository).
|
||||
|
||||
This test is marked as 'slow' and can be skipped in regular test runs:
|
||||
pytest -m "not slow"
|
||||
"""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=1000, test_name="Sync 1000 files (large repository)"
|
||||
)
|
||||
|
||||
# Basic assertions
|
||||
# Note: May be slightly more than 1000 due to OS-generated files
|
||||
assert results["new_files"] >= 1000
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skip
|
||||
async def test_benchmark_resync_no_changes(app_config, project_config, config_manager):
|
||||
"""Benchmark: Re-sync with no changes (should be fast).
|
||||
|
||||
This tests the performance of scanning files when nothing has changed,
|
||||
which is important for cloud restarts.
|
||||
"""
|
||||
project_dir = project_config.home
|
||||
num_files = 100
|
||||
|
||||
# First sync
|
||||
print(f"\nFirst sync of {num_files} files...")
|
||||
await generate_benchmark_files(project_dir, num_files)
|
||||
|
||||
from basic_memory.repository import ProjectRepository
|
||||
from basic_memory import db
|
||||
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
projects = await project_repository.find_all()
|
||||
if projects:
|
||||
project = projects[0]
|
||||
else:
|
||||
project = await project_repository.create(
|
||||
{
|
||||
"name": project_config.name,
|
||||
"path": str(project_config.home),
|
||||
"is_active": True,
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
|
||||
# Initialize search index
|
||||
await sync_service.search_service.init_search_index()
|
||||
|
||||
await sync_service.sync(project_dir, project_name=project.name)
|
||||
|
||||
# Second sync (no changes)
|
||||
print("\nRe-sync with no changes...")
|
||||
resync_start = time.time()
|
||||
report = await sync_service.sync(project_dir, project_name=project.name)
|
||||
resync_duration = time.time() - resync_start
|
||||
|
||||
print(f"\n{'-' * 70}")
|
||||
print("RE-SYNC RESULTS (no changes):")
|
||||
print(f"{'-' * 70}")
|
||||
print(f"Files scanned: {num_files}")
|
||||
print(f"Changes detected: {report.total}")
|
||||
print(f" New: {len(report.new)}")
|
||||
print(f" Modified: {len(report.modified)}")
|
||||
print(f" Deleted: {len(report.deleted)}")
|
||||
print(f" Moved: {len(report.moves)}")
|
||||
print(f"Duration: {resync_duration:.2f}s")
|
||||
print(f"Files/sec: {num_files / resync_duration:.1f}")
|
||||
|
||||
# Debug: Show what changed
|
||||
if report.total > 0:
|
||||
print("\n⚠️ UNEXPECTED CHANGES DETECTED:")
|
||||
if report.new:
|
||||
print(f" New files ({len(report.new)}): {list(report.new)[:5]}")
|
||||
if report.modified:
|
||||
print(f" Modified files ({len(report.modified)}): {list(report.modified)[:5]}")
|
||||
if report.deleted:
|
||||
print(f" Deleted files ({len(report.deleted)}): {list(report.deleted)[:5]}")
|
||||
if report.moves:
|
||||
print(f" Moved files ({len(report.moves)}): {dict(list(report.moves.items())[:5])}")
|
||||
|
||||
print(f"{'=' * 70}\n")
|
||||
|
||||
# Should be no changes
|
||||
assert report.total == 0, (
|
||||
f"Expected no changes but got {report.total}: new={len(report.new)}, modified={len(report.modified)}, deleted={len(report.deleted)}, moves={len(report.moves)}"
|
||||
)
|
||||
assert len(report.new) == 0
|
||||
assert len(report.modified) == 0
|
||||
assert len(report.deleted) == 0
|
||||
@@ -9,17 +9,19 @@ from basic_memory.mcp.async_client import create_client
|
||||
|
||||
|
||||
def test_create_client_uses_asgi_when_no_remote_env():
|
||||
"""Test that create_client uses ASGI transport when BASIC_MEMORY_USE_REMOTE_API is not set."""
|
||||
# Ensure env vars are not set (pop if they exist)
|
||||
"""Test that create_client uses ASGI transport when cloud mode is disabled."""
|
||||
# Ensure env vars are not set and config cloud_mode is False
|
||||
with patch.dict("os.environ", clear=False):
|
||||
os.environ.pop("BASIC_MEMORY_USE_REMOTE_API", None)
|
||||
os.environ.pop("BASIC_MEMORY_CLOUD_MODE", None)
|
||||
|
||||
client = create_client()
|
||||
# Also patch the config's cloud_mode to ensure it's False
|
||||
with patch.object(ConfigManager().config, "cloud_mode", False):
|
||||
client = create_client()
|
||||
|
||||
assert isinstance(client, AsyncClient)
|
||||
assert isinstance(client._transport, ASGITransport)
|
||||
assert str(client.base_url) == "http://test"
|
||||
assert isinstance(client, AsyncClient)
|
||||
assert isinstance(client._transport, ASGITransport)
|
||||
assert str(client.base_url) == "http://test"
|
||||
|
||||
|
||||
def test_create_client_uses_http_when_cloud_mode_env_set():
|
||||
@@ -37,16 +39,18 @@ def test_create_client_uses_http_when_cloud_mode_env_set():
|
||||
|
||||
def test_create_client_configures_extended_timeouts():
|
||||
"""Test that create_client configures 30-second timeouts for long operations."""
|
||||
# Ensure env vars are not set (pop if they exist)
|
||||
# Ensure env vars are not set and config cloud_mode is False
|
||||
with patch.dict("os.environ", clear=False):
|
||||
os.environ.pop("BASIC_MEMORY_USE_REMOTE_API", None)
|
||||
os.environ.pop("BASIC_MEMORY_CLOUD_MODE", None)
|
||||
|
||||
client = create_client()
|
||||
# Also patch the config's cloud_mode to ensure it's False
|
||||
with patch.object(ConfigManager().config, "cloud_mode", False):
|
||||
client = create_client()
|
||||
|
||||
# Verify timeout configuration
|
||||
assert isinstance(client.timeout, Timeout)
|
||||
assert client.timeout.connect == 10.0 # 10 seconds for connection
|
||||
assert client.timeout.read == 30.0 # 30 seconds for reading
|
||||
assert client.timeout.write == 30.0 # 30 seconds for writing
|
||||
assert client.timeout.pool == 30.0 # 30 seconds for pool
|
||||
# Verify timeout configuration
|
||||
assert isinstance(client.timeout, Timeout)
|
||||
assert client.timeout.connect == 10.0 # 10 seconds for connection
|
||||
assert client.timeout.read == 30.0 # 30 seconds for reading
|
||||
assert client.timeout.write == 30.0 # 30 seconds for writing
|
||||
assert client.timeout.pool == 30.0 # 30 seconds for pool
|
||||
|
||||
@@ -218,9 +218,13 @@ async def test_get_resource_entities(client, project_config, entity_repository,
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_resource_entities_pagination(
|
||||
client, project_config, entity_repository, project_url
|
||||
client, project_config, entity_repository, project_url, db_backend
|
||||
):
|
||||
"""Test getting content by permalink match."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip(
|
||||
"Pagination differs: relations expand to multiple entities, ordering is undefined"
|
||||
)
|
||||
# Create entity
|
||||
content1 = "# Test Content\n"
|
||||
data = {
|
||||
|
||||
@@ -48,7 +48,7 @@ async def test_resolve_identifier_not_found(client: AsyncClient, v2_project_url)
|
||||
response = await client.post(f"{v2_project_url}/knowledge/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "Could not resolve identifier" in response.json()["detail"]
|
||||
assert "Entity not found" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
@@ -8,6 +8,7 @@ from httpx import AsyncClient
|
||||
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.schemas.project_info import ProjectItem, ProjectStatusResponse
|
||||
from basic_memory.schemas.v2 import ProjectResolveResponse
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -249,3 +250,85 @@ async def test_update_project_active_status(
|
||||
assert response.status_code == 200
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
assert status_response.status == "success"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_by_name(client: AsyncClient, test_project: Project, v2_projects_url):
|
||||
"""Test resolving a project by name returns correct project ID."""
|
||||
resolve_data = {"identifier": test_project.name}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 200
|
||||
resolved = ProjectResolveResponse.model_validate(response.json())
|
||||
assert resolved.project_id == test_project.id
|
||||
assert resolved.name == test_project.name
|
||||
assert resolved.path == test_project.path
|
||||
assert resolved.is_default == (test_project.is_default or False)
|
||||
# Resolution method could be "name" or "permalink" depending on whether name == permalink
|
||||
assert resolved.resolution_method in ["name", "permalink"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_by_permalink(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Test resolving a project by permalink returns correct project ID."""
|
||||
# Assume test_project.name can be converted to permalink
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
project_permalink = generate_permalink(test_project.name)
|
||||
resolve_data = {"identifier": project_permalink}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 200
|
||||
resolved = ProjectResolveResponse.model_validate(response.json())
|
||||
assert resolved.project_id == test_project.id
|
||||
assert resolved.name == test_project.name
|
||||
# Resolution method could be "name" or "permalink" depending on implementation
|
||||
assert resolved.resolution_method in ["name", "permalink"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_by_id(client: AsyncClient, test_project: Project, v2_projects_url):
|
||||
"""Test resolving a project by ID string returns correct project ID."""
|
||||
resolve_data = {"identifier": str(test_project.id)}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 200
|
||||
resolved = ProjectResolveResponse.model_validate(response.json())
|
||||
assert resolved.project_id == test_project.id
|
||||
assert resolved.name == test_project.name
|
||||
assert resolved.resolution_method == "id"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_case_insensitive(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Test resolving a project by name is case-insensitive."""
|
||||
resolve_data = {"identifier": test_project.name.upper()}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 200
|
||||
resolved = ProjectResolveResponse.model_validate(response.json())
|
||||
assert resolved.project_id == test_project.id
|
||||
assert resolved.name == test_project.name
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_not_found(client: AsyncClient, v2_projects_url):
|
||||
"""Test resolving a non-existent project returns 404."""
|
||||
resolve_data = {"identifier": "nonexistent-project"}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "not found" in response.json()["detail"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_project_empty_identifier(client: AsyncClient, v2_projects_url):
|
||||
"""Test resolving with empty identifier returns 422."""
|
||||
resolve_data = {"identifier": ""}
|
||||
response = await client.post(f"{v2_projects_url}/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 422 # Validation error
|
||||
|
||||
+28
-1
@@ -1,5 +1,8 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import AsyncGenerator
|
||||
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from fastapi import FastAPI
|
||||
from httpx import AsyncClient, ASGITransport
|
||||
@@ -8,7 +11,31 @@ from basic_memory.api.app import app as fastapi_app
|
||||
from basic_memory.deps import get_project_config, get_engine_factory, get_app_config
|
||||
|
||||
|
||||
@pytest_asyncio.fixture(autouse=True)
|
||||
@pytest.fixture(autouse=True)
|
||||
def isolated_home(tmp_path, monkeypatch) -> Path:
|
||||
"""Isolate tests from user's HOME directory.
|
||||
|
||||
This prevents tests from reading/writing to ~/.basic-memory/.bmignore
|
||||
or other user-specific configuration.
|
||||
|
||||
Sets BASIC_MEMORY_HOME to tmp_path directly so the default project
|
||||
writes files to tmp_path, which is where tests expect to find them.
|
||||
"""
|
||||
# Clear config cache to ensure fresh config for each test
|
||||
from basic_memory import config as config_module
|
||||
|
||||
config_module._CONFIG_CACHE = None
|
||||
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
if os.name == "nt":
|
||||
monkeypatch.setenv("USERPROFILE", str(tmp_path))
|
||||
# Set to tmp_path directly (not tmp_path/basic-memory) so default project
|
||||
# home is tmp_path - tests expect to find imported files there
|
||||
monkeypatch.setenv("BASIC_MEMORY_HOME", str(tmp_path))
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def app(app_config, project_config, engine_factory, test_config, aiolib) -> FastAPI:
|
||||
"""Create test FastAPI application."""
|
||||
app = fastapi_app
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Regression tests for CLI command exit behavior.
|
||||
|
||||
These tests verify that CLI commands exit cleanly without hanging,
|
||||
which was a bug fixed in the database initialization refactor.
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def test_bm_version_exits_cleanly():
|
||||
"""Test that 'bm --version' exits cleanly within timeout."""
|
||||
# Use uv run to ensure correct environment
|
||||
result = subprocess.run(
|
||||
["uv", "run", "bm", "--version"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10,
|
||||
cwd=Path(__file__).parent.parent.parent, # Project root
|
||||
)
|
||||
assert result.returncode == 0
|
||||
assert "Basic Memory version:" in result.stdout
|
||||
|
||||
|
||||
def test_bm_help_exits_cleanly():
|
||||
"""Test that 'bm --help' exits cleanly within timeout."""
|
||||
result = subprocess.run(
|
||||
["uv", "run", "bm", "--help"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10,
|
||||
cwd=Path(__file__).parent.parent.parent,
|
||||
)
|
||||
assert result.returncode == 0
|
||||
assert "Basic Memory" in result.stdout
|
||||
|
||||
|
||||
def test_bm_tool_help_exits_cleanly():
|
||||
"""Test that 'bm tool --help' exits cleanly within timeout."""
|
||||
result = subprocess.run(
|
||||
["uv", "run", "bm", "tool", "--help"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10,
|
||||
cwd=Path(__file__).parent.parent.parent,
|
||||
)
|
||||
assert result.returncode == 0
|
||||
assert "tool" in result.stdout.lower()
|
||||
@@ -0,0 +1,85 @@
|
||||
"""Test that CLI tool commands exit cleanly without hanging.
|
||||
|
||||
This test ensures that CLI commands properly clean up database connections
|
||||
on exit, preventing process hangs. See GitHub issue for details.
|
||||
|
||||
The issue occurs when:
|
||||
1. ensure_initialization() calls asyncio.run(initialize_app())
|
||||
2. initialize_app() creates global database connections via db.get_or_create_db()
|
||||
3. When asyncio.run() completes, the event loop closes
|
||||
4. But the global database engine holds async connections that prevent clean exit
|
||||
5. Process hangs indefinitely
|
||||
|
||||
The fix ensures db.shutdown_db() is called before asyncio.run() returns.
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
class TestCLIToolExit:
|
||||
"""Test that CLI tool commands exit cleanly."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"command",
|
||||
[
|
||||
["tool", "--help"],
|
||||
["tool", "write-note", "--help"],
|
||||
["tool", "read-note", "--help"],
|
||||
["tool", "search-notes", "--help"],
|
||||
["tool", "build-context", "--help"],
|
||||
],
|
||||
)
|
||||
def test_cli_command_exits_cleanly(self, command: list[str]):
|
||||
"""Test that CLI commands exit without hanging.
|
||||
|
||||
Each command should complete within the timeout without requiring
|
||||
manual termination (Ctrl+C).
|
||||
"""
|
||||
full_command = [sys.executable, "-m", "basic_memory.cli.main"] + command
|
||||
|
||||
try:
|
||||
result = subprocess.run(
|
||||
full_command,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10.0, # 10 second timeout - commands should complete in ~2s
|
||||
)
|
||||
# Command should exit with code 0 for --help
|
||||
assert result.returncode == 0, f"Command failed: {result.stderr}"
|
||||
except subprocess.TimeoutExpired:
|
||||
pytest.fail(
|
||||
f"Command '{' '.join(command)}' hung and did not exit within timeout. "
|
||||
"This indicates database connections are not being cleaned up properly."
|
||||
)
|
||||
|
||||
def test_ensure_initialization_exits_cleanly(self):
|
||||
"""Test that ensure_initialization doesn't cause process hang.
|
||||
|
||||
This test directly tests the initialization function that's called
|
||||
by CLI commands, ensuring it cleans up database connections properly.
|
||||
"""
|
||||
code = """
|
||||
import asyncio
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.services.initialization import ensure_initialization
|
||||
|
||||
app_config = ConfigManager().config
|
||||
ensure_initialization(app_config)
|
||||
print("OK")
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", code],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10.0,
|
||||
)
|
||||
assert "OK" in result.stdout, f"Unexpected output: {result.stdout}"
|
||||
except subprocess.TimeoutExpired:
|
||||
pytest.fail(
|
||||
"ensure_initialization() caused process hang. "
|
||||
"Database connections are not being cleaned up before event loop closes."
|
||||
)
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user