mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 449645b452 | |||
| ec9b2c4046 | |||
| de7f15b7a2 | |||
| 630eeb94ab | |||
| 0eae0e1678 |
@@ -1,5 +1,16 @@
|
||||
# CHANGELOG
|
||||
|
||||
## v0.18.4 (2026-02-12)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- Use global `--header` flag for Tigris consistency on all rclone transactions
|
||||
([`0eae0e1`](https://github.com/basicmachines-co/basic-memory/commit/0eae0e1))
|
||||
- `--header-download` / `--header-upload` only apply to GET/PUT requests, missing S3
|
||||
ListObjectsV2 calls that bisync issues first. Non-US users saw stale edge-cached metadata.
|
||||
- `--header` applies to ALL HTTP transactions (list, download, upload), fixing bisync for
|
||||
users outside the Tigris origin region.
|
||||
|
||||
## v0.18.2 (2026-02-11)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
+2
-2
@@ -6,12 +6,12 @@
|
||||
"url": "https://github.com/basicmachines-co/basic-memory.git",
|
||||
"source": "github"
|
||||
},
|
||||
"version": "0.18.2",
|
||||
"version": "0.18.5",
|
||||
"packages": [
|
||||
{
|
||||
"registryType": "pypi",
|
||||
"identifier": "basic-memory",
|
||||
"version": "0.18.2",
|
||||
"version": "0.18.5",
|
||||
"runtimeHint": "uvx",
|
||||
"runtimeArguments": [
|
||||
{"type": "positional", "value": "basic-memory"},
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""basic-memory - Local-first knowledge management combining Zettelkasten with knowledge graphs"""
|
||||
|
||||
# Package version - updated by release automation
|
||||
__version__ = "0.18.2"
|
||||
__version__ = "0.18.5"
|
||||
|
||||
# API version for FastAPI - independent of package version
|
||||
__api_version__ = "v0"
|
||||
|
||||
@@ -28,11 +28,13 @@ console = Console()
|
||||
MIN_RCLONE_VERSION_EMPTY_DIRS = (1, 64, 0)
|
||||
|
||||
# Tigris edge caching returns stale data for users outside the origin region (iad).
|
||||
# These headers bypass edge cache and force reads/writes against the origin.
|
||||
# --header is rclone's global flag that applies to ALL HTTP transactions (list, download,
|
||||
# upload). This is critical because bisync starts with S3 ListObjectsV2, which is neither
|
||||
# a download nor upload — so --header-download/--header-upload would miss list requests.
|
||||
# See: https://www.tigrisdata.com/docs/objects/consistency/
|
||||
TIGRIS_CONSISTENCY_HEADERS = [
|
||||
"--header-download", "X-Tigris-Consistent: true",
|
||||
"--header-upload", "X-Tigris-Consistent: true",
|
||||
"--header",
|
||||
"X-Tigris-Consistent: true",
|
||||
]
|
||||
|
||||
|
||||
@@ -221,6 +223,9 @@ def project_sync(
|
||||
*TIGRIS_CONSISTENCY_HEADERS,
|
||||
"--filter-from",
|
||||
str(filter_path),
|
||||
# Prevent NUL byte padding on virtual filesystems (e.g. Google Drive File Stream)
|
||||
# See: rclone/rclone#6801
|
||||
"--local-no-preallocate",
|
||||
]
|
||||
|
||||
if verbose:
|
||||
@@ -297,6 +302,9 @@ def project_bisync(
|
||||
str(filter_path),
|
||||
"--workdir",
|
||||
str(state_path),
|
||||
# Prevent NUL byte padding on virtual filesystems (e.g. Google Drive File Stream)
|
||||
# See: rclone/rclone#6801
|
||||
"--local-no-preallocate",
|
||||
]
|
||||
|
||||
# Add --create-empty-src-dirs if rclone version supports it (v1.64+)
|
||||
|
||||
@@ -19,6 +19,15 @@ from basic_memory.repository.metadata_filters import (
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
|
||||
def _strip_nul_from_row(row_data: dict) -> dict:
|
||||
"""Strip NUL bytes from all string values in a row dict.
|
||||
|
||||
Secondary defense: PostgreSQL text columns cannot store \\x00.
|
||||
Primary sanitization happens in SearchService.index_entity_markdown().
|
||||
"""
|
||||
return {k: v.replace("\x00", "") if isinstance(v, str) else v for k, v in row_data.items()}
|
||||
|
||||
|
||||
class PostgresSearchRepository(SearchRepositoryBase):
|
||||
"""PostgreSQL tsvector implementation of search repository.
|
||||
|
||||
@@ -489,7 +498,7 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
for row in search_index_rows:
|
||||
insert_data = row.to_insert(serialize_json=True)
|
||||
insert_data["project_id"] = self.project_id
|
||||
insert_data_list.append(insert_data)
|
||||
insert_data_list.append(_strip_nul_from_row(insert_data))
|
||||
|
||||
# Use upsert to handle race conditions during parallel indexing
|
||||
# ON CONFLICT (permalink, project_id) matches the partial unique index
|
||||
|
||||
@@ -22,6 +22,15 @@ from basic_memory.services import FileService
|
||||
MAX_CONTENT_STEMS_SIZE = 6000
|
||||
|
||||
|
||||
def _strip_nul(value: str) -> str:
|
||||
"""Strip NUL bytes that PostgreSQL text columns cannot store.
|
||||
|
||||
rclone preallocation on virtual filesystems (e.g. Google Drive File Stream)
|
||||
can pad files with \\x00 bytes. See: rclone/rclone#6801
|
||||
"""
|
||||
return value.replace("\x00", "")
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: Entity) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
@@ -297,7 +306,7 @@ class SearchService:
|
||||
content = await self.file_service.read_entity_content(entity)
|
||||
if content:
|
||||
content_stems.append(content)
|
||||
content_snippet = f"{content[:250]}"
|
||||
content_snippet = _strip_nul(content[:250])
|
||||
|
||||
if entity.permalink:
|
||||
content_stems.extend(self._generate_variants(entity.permalink))
|
||||
@@ -309,7 +318,7 @@ class SearchService:
|
||||
if entity_tags:
|
||||
content_stems.extend(entity_tags)
|
||||
|
||||
entity_content_stems = "\n".join(p for p in content_stems if p and p.strip())
|
||||
entity_content_stems = _strip_nul("\n".join(p for p in content_stems if p and p.strip()))
|
||||
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(entity_content_stems) > MAX_CONTENT_STEMS_SIZE: # pragma: no cover
|
||||
@@ -346,8 +355,8 @@ class SearchService:
|
||||
seen_permalinks.add(obs_permalink)
|
||||
|
||||
# Index with parent entity's file path since that's where it's defined
|
||||
obs_content_stems = "\n".join(
|
||||
p for p in self._generate_variants(obs.content) if p and p.strip()
|
||||
obs_content_stems = _strip_nul(
|
||||
"\n".join(p for p in self._generate_variants(obs.content) if p and p.strip())
|
||||
)
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(obs_content_stems) > MAX_CONTENT_STEMS_SIZE: # pragma: no cover
|
||||
@@ -358,7 +367,7 @@ class SearchService:
|
||||
type=SearchItemType.OBSERVATION.value,
|
||||
title=f"{obs.category}: {obs.content[:100]}...",
|
||||
content_stems=obs_content_stems,
|
||||
content_snippet=obs.content,
|
||||
content_snippet=_strip_nul(obs.content),
|
||||
permalink=obs_permalink,
|
||||
file_path=entity.file_path,
|
||||
category=obs.category,
|
||||
@@ -381,8 +390,8 @@ class SearchService:
|
||||
else f"{rel.from_entity.title}"
|
||||
)
|
||||
|
||||
rel_content_stems = "\n".join(
|
||||
p for p in self._generate_variants(relation_title) if p and p.strip()
|
||||
rel_content_stems = _strip_nul(
|
||||
"\n".join(p for p in self._generate_variants(relation_title) if p and p.strip())
|
||||
)
|
||||
rows_to_index.append(
|
||||
SearchIndexRow(
|
||||
|
||||
@@ -8,7 +8,10 @@ from datetime import datetime, timedelta, timezone
|
||||
|
||||
import pytest
|
||||
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.postgres_search_repository import (
|
||||
PostgresSearchRepository,
|
||||
_strip_nul_from_row,
|
||||
)
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
@@ -197,3 +200,22 @@ async def test_postgres_search_repository_reraises_non_tsquery_db_errors(
|
||||
# Use a non-text query so the generated SQL doesn't include to_tsquery(),
|
||||
# ensuring we hit the generic "re-raise other db errors" branch.
|
||||
await repo.search(permalink="docs/anything")
|
||||
|
||||
|
||||
def test_strip_nul_from_row():
|
||||
"""_strip_nul_from_row strips NUL bytes from string values, leaves non-strings alone."""
|
||||
row = {
|
||||
"title": "hello\x00world",
|
||||
"content_stems": "some\x00content\x00here",
|
||||
"content_snippet": "clean",
|
||||
"id": 42,
|
||||
"metadata": None,
|
||||
"created_at": datetime(2024, 1, 1),
|
||||
}
|
||||
result = _strip_nul_from_row(row)
|
||||
assert result["title"] == "helloworld"
|
||||
assert result["content_stems"] == "somecontenthere"
|
||||
assert result["content_snippet"] == "clean"
|
||||
assert result["id"] == 42
|
||||
assert result["metadata"] is None
|
||||
assert result["created_at"] == datetime(2024, 1, 1)
|
||||
|
||||
@@ -7,6 +7,7 @@ from sqlalchemy import text
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.schemas.search import SearchQuery, SearchItemType
|
||||
from basic_memory.services.search_service import _strip_nul
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -968,3 +969,59 @@ async def test_index_entity_multiple_categories_same_content(
|
||||
# Search for the shared content - should find both observations
|
||||
results = await search_service.search(SearchQuery(text="Shared content"))
|
||||
assert len(results) >= 2
|
||||
|
||||
|
||||
# Tests for NUL byte stripping
|
||||
|
||||
|
||||
def test_strip_nul_removes_nul_bytes():
|
||||
"""_strip_nul removes \\x00 from strings."""
|
||||
assert _strip_nul("hello\x00world") == "helloworld"
|
||||
assert _strip_nul("\x00\x00\x00") == ""
|
||||
assert _strip_nul("clean string") == "clean string"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_index_entity_markdown_strips_nul_bytes(search_service, session_maker, test_project):
|
||||
"""Content with NUL bytes should be stripped before indexing.
|
||||
|
||||
rclone preallocation on virtual filesystems (e.g. Google Drive File Stream)
|
||||
can pad files with \\x00 bytes, causing PostgreSQL CharacterNotInRepertoireError.
|
||||
"""
|
||||
from basic_memory.repository import EntityRepository, ObservationRepository
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
|
||||
entity_repo = EntityRepository(session_maker, project_id=test_project.id)
|
||||
obs_repo = ObservationRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
entity_data = {
|
||||
"title": "NUL Test Entity",
|
||||
"entity_type": "note",
|
||||
"entity_metadata": {},
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "test/nul-test.md",
|
||||
"permalink": "test/nul-test",
|
||||
"project_id": test_project.id,
|
||||
"created_at": datetime.now(),
|
||||
"updated_at": datetime.now(),
|
||||
}
|
||||
entity = await entity_repo.create(entity_data)
|
||||
|
||||
# Add observation with NUL bytes
|
||||
await obs_repo.create(
|
||||
{"entity_id": entity.id, "category": "note", "content": "obs with\x00nul bytes"}
|
||||
)
|
||||
entity = await entity_repo.get_by_permalink("test/nul-test")
|
||||
|
||||
# Index with NUL-containing content
|
||||
nul_content = "# NUL Test\x00\x00\nSome content\x00here"
|
||||
await search_service.index_entity(entity, content=nul_content)
|
||||
|
||||
# Verify no NUL bytes in stored search index rows
|
||||
search_repo: SearchRepository = search_service.repository
|
||||
results = await search_repo.search(permalink_match="test/nul-test*")
|
||||
for row in results:
|
||||
if row.content_snippet:
|
||||
assert "\x00" not in row.content_snippet, (
|
||||
f"NUL found in content_snippet for {row.permalink}"
|
||||
)
|
||||
|
||||
@@ -41,15 +41,14 @@ class _Runner:
|
||||
|
||||
|
||||
def _assert_has_consistency_headers(cmd: list[str]) -> None:
|
||||
"""Assert the rclone command includes Tigris consistency headers."""
|
||||
assert "--header-download" in cmd
|
||||
assert "X-Tigris-Consistent: true" in cmd
|
||||
assert "--header-upload" in cmd
|
||||
# Verify upload header value follows --header-upload
|
||||
upload_idx = cmd.index("--header-upload")
|
||||
assert cmd[upload_idx + 1] == "X-Tigris-Consistent: true"
|
||||
download_idx = cmd.index("--header-download")
|
||||
assert cmd[download_idx + 1] == "X-Tigris-Consistent: true"
|
||||
"""Assert the rclone command includes Tigris consistency headers.
|
||||
|
||||
Uses --header (global flag) so the header applies to ALL HTTP transactions,
|
||||
including S3 list requests that bisync issues before any download/upload.
|
||||
"""
|
||||
assert "--header" in cmd
|
||||
header_idx = cmd.index("--header")
|
||||
assert cmd[header_idx + 1] == "X-Tigris-Consistent: true"
|
||||
|
||||
|
||||
def _write_filter_file(tmp_path: Path) -> Path:
|
||||
@@ -512,3 +511,45 @@ def test_supports_create_empty_src_dirs_false_for_unknown_version():
|
||||
|
||||
def test_min_rclone_version_constant():
|
||||
assert MIN_RCLONE_VERSION_EMPTY_DIRS == (1, 64, 0)
|
||||
|
||||
|
||||
def test_project_sync_includes_no_preallocate_flag(tmp_path):
|
||||
"""Sync command includes --local-no-preallocate to prevent NUL byte padding."""
|
||||
runner = _Runner(returncode=0)
|
||||
filter_path = _write_filter_file(tmp_path)
|
||||
project = SyncProject(name="research", path="/research", local_sync_path="/tmp/research")
|
||||
|
||||
project_sync(
|
||||
project,
|
||||
"my-bucket",
|
||||
run=runner,
|
||||
is_installed=lambda: True,
|
||||
filter_path=filter_path,
|
||||
)
|
||||
|
||||
cmd, _ = runner.calls[0]
|
||||
assert "--local-no-preallocate" in cmd
|
||||
|
||||
|
||||
def test_project_bisync_includes_no_preallocate_flag(tmp_path):
|
||||
"""Bisync command includes --local-no-preallocate to prevent NUL byte padding."""
|
||||
runner = _Runner(returncode=0)
|
||||
filter_path = _write_filter_file(tmp_path)
|
||||
state_path = tmp_path / "state"
|
||||
project = SyncProject(
|
||||
name="research", path="app/data/research", local_sync_path="/tmp/research"
|
||||
)
|
||||
|
||||
project_bisync(
|
||||
project,
|
||||
"my-bucket",
|
||||
run=runner,
|
||||
is_installed=lambda: True,
|
||||
version=(1, 64, 2),
|
||||
filter_path=filter_path,
|
||||
state_path=state_path,
|
||||
is_initialized=lambda _name: True,
|
||||
)
|
||||
|
||||
cmd, _ = runner.calls[0]
|
||||
assert "--local-no-preallocate" in cmd
|
||||
|
||||
Reference in New Issue
Block a user