mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
refactor markdown parsing models
This commit is contained in:
@@ -1,3 +1,5 @@
|
||||
"""Models for the markdown parser."""
|
||||
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
@@ -26,23 +28,19 @@ class Relation(BaseModel):
|
||||
|
||||
|
||||
class EntityFrontmatter(BaseModel):
|
||||
"""Frontmatter metadata for an entity."""
|
||||
"""Required frontmatter fields for an entity."""
|
||||
|
||||
type: str
|
||||
id: str
|
||||
created: datetime
|
||||
modified: datetime
|
||||
tags: List[str]
|
||||
status: Optional[str] = None
|
||||
version: Optional[int] = None
|
||||
priority: Optional[str] = None
|
||||
domain: Optional[str] = None
|
||||
maturity: Optional[str] = None
|
||||
owner: Optional[str] = None
|
||||
review_interval: Optional[str] = None
|
||||
last_reviewed: Optional[datetime] = None
|
||||
confidence: Optional[str] = None
|
||||
aliases: Optional[List[str]] = None
|
||||
|
||||
|
||||
class EntityMetadata(BaseModel):
|
||||
"""Optional metadata fields for an entity (backmatter)."""
|
||||
|
||||
metadata: Dict[str, Any] = {}
|
||||
|
||||
|
||||
class EntityContent(BaseModel):
|
||||
@@ -53,11 +51,11 @@ class EntityContent(BaseModel):
|
||||
observations: List[Observation] = []
|
||||
relations: List[Relation] = []
|
||||
context: Optional[str] = None
|
||||
metadata: Dict[str, Any] = {}
|
||||
|
||||
|
||||
class Entity(BaseModel):
|
||||
"""Complete entity combining frontmatter and content."""
|
||||
"""Complete entity combining frontmatter, content, and metadata."""
|
||||
|
||||
frontmatter: EntityFrontmatter
|
||||
content: EntityContent
|
||||
content: EntityContent
|
||||
metadata: EntityMetadata = EntityMetadata()
|
||||
@@ -115,10 +115,16 @@ class EntityParser:
|
||||
|
||||
def _parse_metadata_line(self, line: str) -> Optional[Tuple[str, str]]:
|
||||
"""Parse a metadata line into key-value pair."""
|
||||
if ":" not in line:
|
||||
if not line.strip():
|
||||
return None
|
||||
try:
|
||||
# Split on first colon
|
||||
if ":" not in line:
|
||||
return None
|
||||
key, value = line.split(":", 1)
|
||||
return key.strip(), value.strip()
|
||||
except ValueError:
|
||||
return None
|
||||
key, value = line.split(":", 1)
|
||||
return key.strip(), value.strip()
|
||||
|
||||
def parse_file(self, path: Path, encoding: str = "utf-8") -> Entity:
|
||||
"""Parse an entity markdown file."""
|
||||
@@ -157,18 +163,11 @@ class EntityParser:
|
||||
list_item_level = None
|
||||
base_list_level = None
|
||||
|
||||
# Track metadata continuation state
|
||||
current_meta_key = None
|
||||
current_meta_value = []
|
||||
# Track metadata state
|
||||
in_metadata_para = False
|
||||
|
||||
for token in tokens:
|
||||
if token.type == "heading_open":
|
||||
# Handle any pending metadata
|
||||
if current_meta_key:
|
||||
metadata[current_meta_key] = " ".join(current_meta_value).strip()
|
||||
current_meta_key = None
|
||||
current_meta_value = []
|
||||
|
||||
if token.tag == "h1":
|
||||
current_section = "title"
|
||||
elif token.tag == "h2":
|
||||
@@ -186,24 +185,12 @@ class EntityParser:
|
||||
current_section = "description"
|
||||
elif current_section == "section_name":
|
||||
section = content.lower()
|
||||
if section == "description":
|
||||
description_tokens = []
|
||||
current_section = section
|
||||
elif current_section == "description":
|
||||
description_tokens.append(token)
|
||||
elif current_section == "metadata":
|
||||
# Parse metadata line
|
||||
if ":" in content:
|
||||
# Save previous key if exists
|
||||
if current_meta_key:
|
||||
metadata[current_meta_key] = " ".join(current_meta_value).strip()
|
||||
# Start new key
|
||||
key, value = content.split(":", 1)
|
||||
current_meta_key = key.strip()
|
||||
current_meta_value = [value.strip()]
|
||||
elif content.strip() and current_meta_key:
|
||||
# Continue previous value
|
||||
current_meta_value.append(content.strip())
|
||||
if parsed := self._parse_metadata_line(content):
|
||||
metadata[parsed[0]] = parsed[1]
|
||||
elif in_list_item:
|
||||
list_item_tokens.append(token)
|
||||
|
||||
@@ -231,14 +218,16 @@ class EntityParser:
|
||||
in_list_item = False
|
||||
list_item_tokens = []
|
||||
|
||||
elif token.type == "paragraph_open":
|
||||
in_metadata_para = current_section == "metadata"
|
||||
|
||||
elif token.type == "paragraph_close":
|
||||
in_metadata_para = False
|
||||
|
||||
# Handle any remaining description
|
||||
if current_section == "description" and description_tokens:
|
||||
description = " ".join(t.content for t in description_tokens)
|
||||
|
||||
# Handle any remaining metadata
|
||||
if current_meta_key:
|
||||
metadata[current_meta_key] = " ".join(current_meta_value).strip()
|
||||
|
||||
# Create entity
|
||||
content_data = EntityContent(
|
||||
title=title,
|
||||
|
||||
Reference in New Issue
Block a user