fix some parser.py tests

This commit is contained in:
phernandez
2024-12-21 15:40:25 -06:00
parent 0c8f729769
commit 252da2af94
+110 -33
View File
@@ -3,7 +3,7 @@
import logging
import re
from pathlib import Path
from typing import Optional
from typing import Optional, List, Tuple, Dict
import frontmatter
from markdown_it import MarkdownIt
@@ -33,11 +33,19 @@ class EntityParser:
if not content.strip():
return None
# Check for unclosed bracket first
if "[" in content and "]" not in content:
raise ParseError("unclosed category")
# Parse category [type]
match = re.match(r"^\s*(?:-\s*)?\[([^\]]+)\](.*)", content)
match = re.match(r"^\s*(?:-\s*)?\[([^\]]*)\](.*)", content)
if not match:
return None
raise ParseError("missing category")
category = match.group(1).strip()
if not category:
return None
content = match.group(2).strip()
# Parse tags and content
@@ -63,6 +71,8 @@ class EntityParser:
content = content[:ctx_start].strip()
return Observation(category=category, content=content, tags=tags, context=context)
except ParseError:
raise
except Exception:
logger.exception("Failed to parse observation: %s", content)
return None
@@ -73,10 +83,14 @@ class EntityParser:
if not content.strip():
return None
# Check for unclosed [[
if "[[" in content and "]]" not in content:
raise ParseError("missing ]]")
# Find the link
match = re.search(r"\[\[([^\]]+)\]\]", content)
if not match:
return None
raise ParseError("missing [[")
target = match.group(1).strip()
before_link = content[: match.start()].strip(" -")
@@ -93,10 +107,19 @@ class EntityParser:
context = after_link[1:-1].strip()
return Relation(target=target, type=rel_type, context=context)
except ParseError:
raise
except Exception:
logger.exception("Failed to parse relation: %s", content)
return None
def _parse_metadata_line(self, line: str) -> Optional[Tuple[str, str]]:
"""Parse a metadata line into key-value pair."""
if ":" not in line:
return None
key, value = line.split(":", 1)
return key.strip(), value.strip()
def parse_file(self, path: Path, encoding: str = "utf-8") -> Entity:
"""Parse an entity markdown file."""
if not path.exists():
@@ -111,8 +134,8 @@ class EntityParser:
# Handle frontmatter
metadata = dict(post.metadata)
if isinstance(metadata.get("tags"), str):
metadata["tags"] = [t.strip() for t in metadata["tags"].split(",")] # pyright: ignore [reportAttributeAccessIssue]
frontmatter_data = EntityFrontmatter(**metadata) # pyright: ignore [reportArgumentType]
metadata["tags"] = [t.strip() for t in metadata["tags"].split(",")]
frontmatter_data = EntityFrontmatter(**metadata)
# Parse markdown
tokens = self.md.parse(post.content)
@@ -120,47 +143,101 @@ class EntityParser:
# State for parsing
title = ""
description = ""
observations = []
relations = []
observations: List[Observation] = []
relations: List[Relation] = []
context = ""
metadata = {}
current_section = None
current_list = []
in_list = False
description_tokens = []
# Track list item state
in_list_item = False
list_item_tokens = []
list_item_level = None
base_list_level = None
# Track metadata continuation state
current_meta_key = None
current_meta_value = []
# Process tokens
for token in tokens:
if token.type == "heading_open":
# Handle any pending metadata
if current_meta_key:
metadata[current_meta_key] = " ".join(current_meta_value).strip()
current_meta_key = None
current_meta_value = []
if token.tag == "h1":
current_section = "title"
elif token.tag == "h2":
# Handle previous section
if current_section == "description":
description = " ".join(t.content for t in description_tokens)
current_section = "section_name"
description_tokens = []
elif token.type == "inline":
content = token.content.strip()
if current_section == "title":
title = token.content
current_section = None
title = content
current_section = "description"
elif current_section == "section_name":
section = token.content.lower()
section = content.lower()
if section == "description":
description_tokens = []
current_section = section
if section in ["observations", "relations"]:
in_list = False
current_list = []
elif current_section == "description":
description = token.content
elif current_section == "observations" and token.content.strip():
if obs := self._parse_observation(token.content):
observations.append(obs)
elif current_section == "relations" and token.content.strip():
if rel := self._parse_relation(token.content):
relations.append(rel)
elif current_section == "context":
context = token.content
elif token.type == "bullet_list_open":
in_list = True
elif token.type == "bullet_list_close":
in_list = False
description_tokens.append(token)
elif current_section == "metadata":
# Parse metadata line
if ":" in content:
# Save previous key if exists
if current_meta_key:
metadata[current_meta_key] = " ".join(current_meta_value).strip()
# Start new key
key, value = content.split(":", 1)
current_meta_key = key.strip()
current_meta_value = [value.strip()]
elif content.strip() and current_meta_key:
# Continue previous value
current_meta_value.append(content.strip())
elif in_list_item:
list_item_tokens.append(token)
elif token.type == "list_item_open":
in_list_item = True
list_item_level = token.level
if base_list_level is None:
base_list_level = token.level
list_item_tokens = []
elif token.type == "list_item_close" and in_list_item:
# Only process top-level items
if base_list_level is None or list_item_level == base_list_level:
item_content = " ".join(t.content for t in list_item_tokens)
try:
if current_section == "observations":
if obs := self._parse_observation(item_content):
observations.append(obs)
elif current_section == "relations":
if rel := self._parse_relation(item_content):
relations.append(rel)
except ParseError:
# Skip malformed items
pass
in_list_item = False
list_item_tokens = []
# Handle any remaining description
if current_section == "description" and description_tokens:
description = " ".join(t.content for t in description_tokens)
# Handle any remaining metadata
if current_meta_key:
metadata[current_meta_key] = " ".join(current_meta_value).strip()
# Create entity
content_data = EntityContent(
@@ -179,4 +256,4 @@ class EntityParser:
return self.parse_file(path, encoding="utf-16")
raise ParseError(f"Failed to read {path} with encoding {encoding}: {str(e)}")
except Exception as e:
raise ParseError(f"Failed to parse {path}: {str(e)}") from e
raise ParseError(f"Failed to parse {path}: {str(e)}") from e