Core: Remove inline file-top metadata and migrate to centralized JSON (#349)

This commit is contained in:
Minseok Song
2026-01-27 18:21:56 +09:00
committed by GitHub
parent 0567a6d141
commit 5d44c6a3b4
9 changed files with 450 additions and 186 deletions
+18 -4
View File
@@ -182,15 +182,29 @@ def evaluate_command(
except ValueError:
display_path = str(file_path).replace("\\", "/")
# Read file to get issues for reference
# Read centralized metadata to get issues for reference
try:
with open(file_path, "r", encoding="utf-8") as f:
content = f.read()
from co_op_translator.utils.common.metadata_utils import (
read_text_metadata_for_source,
extract_metadata_from_content,
)
metadata = extract_metadata_from_content(content)
trans_path = Path(file_path)
lang_dir = root_path / "translations" / language_code
try:
rel = trans_path.resolve().relative_to(lang_dir)
orig_path = root_path / rel
except Exception:
orig_path = None
metadata = {}
if orig_path is not None:
metadata = read_text_metadata_for_source(lang_dir, orig_path)
if not metadata:
with open(file_path, "r", encoding="utf-8") as f:
content = f.read()
metadata = extract_metadata_from_content(content)
issues = []
if metadata and "evaluation" in metadata:
# Only check issues field
@@ -10,10 +10,7 @@ import logging
from pathlib import Path
from .markdown_translator import MarkdownTranslator
from co_op_translator.utils.common.metadata_utils import (
calculate_file_hash,
add_notebook_metadata,
)
from co_op_translator.utils.common.metadata_utils import calculate_file_hash
logger = logging.getLogger(__name__)
@@ -164,14 +161,6 @@ class JupyterNotebookTranslator:
notebook["cells"].append(disclaimer_cell)
logger.debug(f"Added disclaimer cell to {notebook_path.name}")
# Add coopTranslator metadata to the notebook
notebook_path = Path(notebook_path)
notebook = add_notebook_metadata(
notebook, notebook_path, language_code, self.root_dir
)
logger.debug(f"Added coopTranslator metadata to {notebook_path.name}")
# Return the modified notebook as JSON string
return json.dumps(notebook, ensure_ascii=False, indent=1)
@@ -8,7 +8,8 @@ from co_op_translator.config.llm_config.provider import LLMProvider
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
extract_content_without_metadata,
format_metadata_comment,
read_text_metadata_for_source,
save_text_metadata_for_source,
)
from co_op_translator.utils.llm.markdown_utils import (
generate_evaluation_prompt,
@@ -254,11 +255,46 @@ class MarkdownEvaluator(ABC):
with open(translated_file, "r", encoding="utf-8") as f:
translated_content = f.read()
# Extract existing metadata from translated file
metadata = extract_metadata_from_content(translated_content)
if not metadata:
logger.warning(f"No metadata found in {translated_file}")
return {}, False
# Determine language directory and original file path
translated_file = Path(translated_file)
# Ascend to find the language directory containing .co-op-translator.json
def _find_lang_dir(path: Path) -> Path | None:
cur = path.parent
while True:
meta = cur / ".co-op-translator.json"
if meta.exists():
return cur
parent = cur.parent
if parent == cur:
return None
cur = parent
lang_dir = _find_lang_dir(translated_file)
if lang_dir is None:
# Fallback: try to locate 'translations/<lang>' in the path
parts = translated_file.resolve().parts
try:
idx = parts.index("translations")
lang_dir = Path(*parts[: idx + 2])
except ValueError:
lang_dir = translated_file.parent
try:
rel = translated_file.resolve().relative_to(lang_dir)
original_file = (self.root_dir or Path.cwd()) / rel
except Exception:
# As last resort, attempt to extract from legacy inline metadata
metadata_legacy = extract_metadata_from_content(translated_content)
source_file = (
metadata_legacy.get("source_file") if metadata_legacy else None
)
if not source_file:
logger.warning(
f"Unable to resolve original path for {translated_file}"
)
return {}, False
original_file = (self.root_dir or Path.cwd()) / source_file
# Initialize evaluation results
rule_based_result = {"confidence_score": 1.0, "issues_found": []}
@@ -371,29 +407,26 @@ class MarkdownEvaluator(ABC):
rule_based_result, chunk_evaluations
)
# Update metadata with evaluation
metadata["evaluation"] = evaluation_result
# Format updated metadata
metadata_comment = format_metadata_comment(metadata)
# Update the file with new metadata
# Extract the actual content without metadata
content_without_metadata = extract_content_without_metadata(
translated_content
)
# Create updated content with new metadata followed by original content
updated_content = metadata_comment
if content_without_metadata.strip():
updated_content += content_without_metadata
# Write back to file
with open(translated_file, "w", encoding="utf-8") as f:
f.write(updated_content)
logger.info(f"Updated evaluation metadata for {translated_file}")
return evaluation_result, True
# Save evaluation into centralized language metadata
language_code = lang_dir.name if lang_dir else ""
extra_fields = {"evaluation": evaluation_result}
try:
save_text_metadata_for_source(
lang_dir,
original_file,
language_code,
root_dir=self.root_dir,
extra_fields=extra_fields,
)
logger.info(
f"Updated evaluation metadata for {translated_file} in centralized JSON"
)
return evaluation_result, True
except Exception as e:
logger.error(
f"Failed to save evaluation metadata for {translated_file}: {e}"
)
return {}, False
except Exception as e:
logger.error(f"Error evaluating {translated_file}: {e}")
@@ -10,6 +10,7 @@ from pathlib import PurePosixPath
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
remove_image_metadata,
remove_text_metadata_for_source,
)
from co_op_translator.config.constants import (
SUPPORTED_MARKDOWN_EXTENSIONS,
@@ -210,29 +211,54 @@ class DirectoryManager:
continue
logger.info(f"Processing translation file: {trans_file}")
# Read translation file and extract metadata using utility
content = trans_file.read_text(encoding="utf-8")
metadata = extract_metadata_from_content(content)
if not metadata:
logger.warning(f"No metadata found in: {trans_file}")
continue
source_file = metadata.get("source_file")
if not source_file:
logger.warning(f"No source_file in metadata: {trans_file}")
continue
original_file = None
# Prefer legacy inline metadata if present to resolve original path robustly
try:
content = trans_file.read_text(encoding="utf-8")
metadata = extract_metadata_from_content(content)
source_file = (
metadata.get("source_file") if metadata else None
)
if source_file:
# Normalize backslashes and construct a proper relative Path
rel_parts = (
str(source_file).replace("\\", "/").split("/")
)
rel_path = Path(*rel_parts)
original_file = self.root_dir / rel_path
except Exception:
# Ignore content read/parse issues; will fallback to relative mapping
pass
normalized_path = str(PurePosixPath(source_file))
original_file = self.root_dir / normalized_path
if original_file is None:
# Fallback: compute original path by relative path from language dir
try:
rel = trans_file.relative_to(translation_dir)
original_file = self.root_dir / rel
except ValueError:
logger.warning(
f"Unable to determine source for: {trans_file}"
)
continue
logger.info(f"Checking original file: {original_file}")
if not original_file.exists():
logger.info(
f"Original file not found, deleting: {trans_file}"
)
trans_file.unlink()
removed_count += 1
logger.info(f"Successfully deleted: {trans_file}")
try:
trans_file.unlink()
removed_count += 1
logger.info(f"Successfully deleted: {trans_file}")
finally:
# Remove centralized metadata entry for this source
try:
remove_text_metadata_for_source(
translation_dir, original_file
)
except Exception:
pass
parent = trans_file.parent
while parent != translation_dir:
@@ -282,28 +308,51 @@ class DirectoryManager:
if not nb_file.exists():
continue
with open(nb_file, "r", encoding="utf-8") as f:
nb_json = json.load(f)
coop_meta = nb_json.get("metadata", {}).get(
"coopTranslator", {}
)
source_file = coop_meta.get("source_file")
if not source_file:
logger.warning(
f"No source_file in notebook metadata: {nb_file}"
original_file = None
# Prefer legacy notebook inline metadata if present
try:
with open(nb_file, "r", encoding="utf-8") as f:
nb_json = json.load(f)
coop_meta = nb_json.get("metadata", {}).get(
"coopTranslator", {}
)
continue
source_file = coop_meta.get("source_file")
if source_file:
rel_parts = (
str(source_file).replace("\\", "/").split("/")
)
rel_path = Path(*rel_parts)
original_file = self.root_dir / rel_path
except Exception:
# Ignore JSON read/parse issues; will fallback to relative mapping
pass
normalized_path = str(PurePosixPath(source_file))
original_file = self.root_dir / normalized_path
if original_file is None:
# Fallback: compute original notebook path by relative path from language dir
try:
rel = nb_file.relative_to(translation_dir)
original_file = self.root_dir / rel
except ValueError:
logger.warning(
f"Unable to determine source for notebook: {nb_file}"
)
continue
if not original_file.exists():
logger.info(
f"Original notebook not found, deleting: {nb_file}"
)
nb_file.unlink()
removed_count += 1
try:
nb_file.unlink()
removed_count += 1
finally:
# Remove centralized metadata entry for this source
try:
remove_text_metadata_for_source(
translation_dir, original_file
)
except Exception:
pass
parent = nb_file.parent
while parent != translation_dir:
@@ -9,7 +9,10 @@ from tqdm import tqdm
from co_op_translator.core.llm.markdown_evaluator import MarkdownEvaluator
from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS
from co_op_translator.utils.common.metadata_utils import extract_metadata_from_content
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
read_text_metadata_for_source,
)
logger = logging.getLogger(__name__)
@@ -98,6 +101,9 @@ class ProjectEvaluator:
bar_format="{l_bar}{bar}| {n_fmt}/{total_fmt} [{elapsed}<{remaining}, {rate_fmt}]",
)
# Precompute language directory for centralized metadata
lang_dir = self.translations_dir / language_code
# Perform rule-based evaluation for all files
for orig_file, trans_file in translation_pairs:
try:
@@ -108,11 +114,10 @@ class ProjectEvaluator:
with open(trans_file, "r", encoding="utf-8") as f:
translated_content = f.read()
# Extract existing metadata
metadata = extract_metadata_from_content(translated_content)
# Prefer centralized metadata; fall back to inline
metadata = read_text_metadata_for_source(lang_dir, orig_file)
if not metadata:
logger.warning(f"No metadata found in {trans_file}")
continue
metadata = extract_metadata_from_content(translated_content)
# Store file info for later LLM evaluation
file_key = str(trans_file)
@@ -399,23 +404,24 @@ class ProjectEvaluator:
translation_pairs = await self._get_translation_pairs(language_code)
low_confidence_translations = []
for _, trans_file in translation_pairs:
# Prefer centralized JSON metadata
lang_dir = self.translations_dir / language_code
for orig_file, trans_file in translation_pairs:
try:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
)
metadata = extract_metadata_from_content(content)
metadata = read_text_metadata_for_source(lang_dir, orig_file)
if not metadata:
# Fallback to inline
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
metadata = extract_metadata_from_content(content)
if metadata and "evaluation" in metadata:
confidence = metadata["evaluation"].get("confidence_score", 1.0)
if confidence < threshold:
low_confidence_translations.append((trans_file, confidence))
except Exception as e:
logger.error(f"Error reading metadata from {trans_file}: {e}")
logger.error(f"Error reading evaluation metadata for {trans_file}: {e}")
return low_confidence_translations
@@ -434,16 +440,15 @@ class ProjectEvaluator:
translation_pairs = await self._get_translation_pairs(language_code)
translations_with_issues = []
for _, trans_file in translation_pairs:
lang_dir = self.translations_dir / language_code
for orig_file, trans_file in translation_pairs:
try:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
)
metadata = extract_metadata_from_content(content)
metadata = read_text_metadata_for_source(lang_dir, orig_file)
if not metadata:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
metadata = extract_metadata_from_content(content)
if metadata and "evaluation" in metadata:
confidence = metadata["evaluation"].get("confidence_score", 1.0)
@@ -453,7 +458,9 @@ class ProjectEvaluator:
(trans_file, confidence, issues)
)
except Exception as e:
logger.error(f"Error reading metadata from {trans_file}: {e}")
logger.error(
f"Error reading evaluation metadata from {trans_file}: {e}"
)
return translations_with_issues
@@ -473,16 +480,15 @@ class ProjectEvaluator:
translation_pairs = await self._get_translation_pairs(language_code)
problematic_translations = []
for _, trans_file in translation_pairs:
lang_dir = self.translations_dir / language_code
for orig_file, trans_file in translation_pairs:
try:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
)
metadata = extract_metadata_from_content(content)
metadata = read_text_metadata_for_source(lang_dir, orig_file)
if not metadata:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
metadata = extract_metadata_from_content(content)
if metadata and "evaluation" in metadata:
confidence = metadata["evaluation"].get("confidence_score", 1.0)
@@ -493,6 +499,8 @@ class ProjectEvaluator:
(trans_file, confidence, issues)
)
except Exception as e:
logger.error(f"Error reading metadata from {trans_file}: {e}")
logger.error(
f"Error reading evaluation metadata from {trans_file}: {e}"
)
return problematic_translations
@@ -266,33 +266,30 @@ class ProjectTranslator:
for trans_file_path, confidence in low_confidence_files:
trans_file = Path(trans_file_path)
# Extract metadata to get original file path
# Compute original file path by relative path from translations/<lang>/
try:
with open(trans_file, "r", encoding="utf-8") as f:
content = f.read()
rel = trans_file.resolve().relative_to(self.translations_dir)
# drop first part (lang code)
if len(rel.parts) < 2:
raise ValueError("Unexpected translation path structure")
lang_code = rel.parts[0]
orig_rel = Path(*rel.parts[1:])
orig_file = (self.root_dir / orig_rel).resolve()
from co_op_translator.utils.common.metadata_utils import (
extract_metadata_from_content,
)
metadata = extract_metadata_from_content(content)
if metadata and "source_file" in metadata:
# Get original file path from metadata
orig_file = self.root_dir / metadata["source_file"]
if orig_file.exists():
files_to_retranslate.append((orig_file, confidence))
else:
error_msg = f"Original source file not found for translation '{trans_file.name}': Expected file '{orig_file}' does not exist. The source may have been moved or deleted."
logger.warning(error_msg)
errors.append(error_msg)
if orig_file.exists():
files_to_retranslate.append((orig_file, confidence))
else:
error_msg = f"Invalid translation metadata in file '{trans_file.name}': Missing source_file information. The file may be corrupted or not generated by Co-op Translator."
error_msg = (
f"Original source file not found for translation '{trans_file.name}': "
f"Expected file '{orig_file}' does not exist. The source may have been moved or deleted."
)
logger.warning(error_msg)
errors.append(error_msg)
except Exception as e:
error_msg = f"Failed to read translation file '{trans_file.name}': {str(e)}. Check file permissions and encoding."
error_msg = (
f"Failed to resolve original file for translation '{trans_file.name}': {str(e)}. "
f"Ensure translations directory structure matches translations/<lang>/..."
)
logger.error(error_msg)
errors.append(error_msg)
@@ -23,6 +23,10 @@ from co_op_translator.utils.common.metadata_utils import (
is_image_up_to_date,
remove_image_metadata,
cleanup_orphan_image_metadata,
save_text_metadata_for_source,
read_text_metadata_for_source,
extract_metadata_from_content,
extract_content_without_metadata,
)
from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS
from co_op_translator.core.llm.markdown_translator import MarkdownTranslator
@@ -166,12 +170,13 @@ class TranslationManager:
handle_empty_document(file_path, output_file)
return str(output_file)
# Perform initial translation attempt
# Perform initial translation attempt (do not embed inline metadata; use centralized JSON instead)
translated_content = await self.markdown_translator.translate_markdown(
document,
language_code,
file_path,
translation_types=self.translation_types,
add_metadata=False,
add_disclaimer=self.add_disclaimer,
)
if not translated_content:
@@ -191,6 +196,7 @@ class TranslationManager:
language_code,
file_path,
translation_types=self.translation_types,
add_metadata=False,
add_disclaimer=self.add_disclaimer,
)
if not translated_content:
@@ -211,6 +217,14 @@ class TranslationManager:
logger.info(
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
# Save centralized text metadata for this source file in the language directory
lang_dir = self.translations_dir / language_code
save_text_metadata_for_source(
lang_dir,
file_path,
language_code,
root_dir=self.root_dir,
)
return str(translated_path)
except Exception as e:
logger.error(f"Failed to write translation to {translated_path}: {e}")
@@ -265,6 +279,14 @@ class TranslationManager:
logger.info(
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
# Save centralized text metadata for this source notebook in the language directory
lang_dir = self.translations_dir / language_code
save_text_metadata_for_source(
lang_dir,
file_path,
language_code,
root_dir=self.root_dir,
)
return str(translated_path)
except Exception as e:
logger.error(f"Failed to write translation to {translated_path}: {e}")
@@ -1186,34 +1208,84 @@ class TranslationManager:
return True
try:
# Handle notebook files differently from markdown files
# Handle notebook files using dedicated helper (centralized JSON preferred inside helper)
if translation_file.suffix.lower() in self.supported_notebook_extensions:
# Use the dedicated notebook metadata comparison function
return not is_notebook_up_to_date(original_file, translation_file)
# Handle markdown files with HTML comment metadata
content = translation_file.read_text(encoding="utf-8")
metadata_match = re.search(
r"<!--\s*CO_OP_TRANSLATOR_METADATA:\s*(.*?)\s*-->",
content,
re.DOTALL,
)
if not metadata_match:
return True
# Determine language directory from translation path
lang_dir = None
try:
metadata = json.loads(metadata_match.group(1))
except json.JSONDecodeError:
rel = translation_file.resolve().relative_to(self.translations_dir)
lang_code = rel.parts[0]
lang_dir = self.translations_dir / lang_code
except Exception:
# Fallback: use the parent directory (may be incorrect for deeply nested paths)
lang_dir = translation_file.parent
# Prefer centralized JSON metadata
metadata = read_text_metadata_for_source(lang_dir, original_file)
if metadata and isinstance(metadata, dict):
stored_hash = metadata.get("original_hash")
if stored_hash:
current_hash = calculate_file_hash(original_file)
return stored_hash != current_hash
# Legacy fallback: read inline HTML comment metadata and migrate
try:
content = translation_file.read_text(encoding="utf-8")
except Exception:
return True
# Determine if content has changed since last translation
original_hash = calculate_file_hash(original_file)
stored_hash = metadata.get("original_hash")
legacy_meta = extract_metadata_from_content(content)
stored_hash = (
legacy_meta.get("original_hash")
if isinstance(legacy_meta, dict)
else None
)
if stored_hash:
# Migrate legacy metadata into centralized JSON
extra_fields = {}
# Preserve original_hash and translation_date from legacy metadata if present
extra_fields["original_hash"] = stored_hash
if "translation_date" in legacy_meta:
extra_fields["translation_date"] = legacy_meta.get(
"translation_date"
)
if not stored_hash:
return True
# Compute language code again for save (if available)
language_code = None
try:
if "rel" not in locals():
rel = translation_file.resolve().relative_to(
self.translations_dir
)
language_code = rel.parts[0]
except Exception:
language_code = None
return stored_hash != original_hash
if language_code:
save_text_metadata_for_source(
lang_dir,
original_file,
language_code,
root_dir=self.root_dir,
extra_fields=extra_fields,
)
# Remove inline metadata from the translated markdown file
cleaned = extract_content_without_metadata(content)
if cleaned != content:
try:
translation_file.write_text(cleaned, encoding="utf-8")
except Exception:
# Non-fatal; continue
pass
current_hash = calculate_file_hash(original_file)
return stored_hash != current_hash
# No metadata available; consider it outdated
return True
except Exception:
return True
@@ -266,7 +266,20 @@ def is_notebook_up_to_date(original_path: Path, translated_path: Path) -> bool:
Returns:
True if translated notebook is up to date, False otherwise
"""
try:
translated_path = Path(translated_path)
if not translated_path.exists():
return False
lang_dir = _find_lang_dir_for_translated_file(translated_path)
if lang_dir is not None:
metadata = read_text_metadata_for_source(lang_dir, original_path)
stored_hash = metadata.get("original_hash")
if stored_hash:
current_hash = calculate_file_hash(Path(original_path))
return stored_hash == current_hash
stored_hash = read_notebook_metadata(translated_path, "coopTranslator").get(
"original_hash"
)
@@ -309,6 +322,23 @@ def add_notebook_metadata(
return notebook
def _find_lang_dir_for_translated_file(translated_path: Path) -> Optional[Path]:
translated_path = Path(translated_path)
current = translated_path.parent
while True:
metadata_file = _get_metadata_file_path(current)
if metadata_file.exists():
return current
parent = current.parent
if parent == current:
break
current = parent
return None
# ============================================================================
# Image-specific metadata utilities
# ============================================================================
@@ -543,3 +573,101 @@ def is_image_up_to_date(
except Exception as e:
logger.debug(f"Error checking image up-to-date status: {e}")
return False
def _normalize_source_key(source: str | Path) -> str:
return str(source).replace("\\", "/")
def load_language_metadata(lang_dir: Path) -> dict:
return _load_lang_metadata(lang_dir)
def save_language_metadata(lang_dir: Path, metadata: dict) -> None:
lock = _get_lock_for_path(_get_metadata_file_path(lang_dir))
with lock:
existing = _load_lang_metadata(lang_dir)
existing.update(metadata)
_save_lang_metadata(lang_dir, existing)
def save_text_metadata_for_source(
lang_dir: Path,
original_file: Path,
language_code: str,
root_dir: Path | None = None,
extra_fields: Optional[dict] = None,
) -> dict:
metadata = create_metadata(original_file, language_code, root_dir)
if extra_fields:
metadata.update(extra_fields)
source_key = metadata.get("source_file")
if not source_key:
return metadata
normalized_key = _normalize_source_key(source_key) # relative, POSIX-style
lock = _get_lock_for_path(_get_metadata_file_path(lang_dir))
with lock:
all_metadata = _load_lang_metadata(lang_dir)
all_metadata[normalized_key] = metadata
_save_lang_metadata(lang_dir, all_metadata)
return metadata
def read_text_metadata_for_source(lang_dir: Path, source_file: str | Path) -> dict:
"""Read text metadata by source path.
Keys are stored as repo-relative, POSIX-style paths. This lookup accepts any
of the following and resolves to the best matching relative key:
- relative POSIX-style path (exact match)
- absolute path (Windows or POSIX): matched by the longest suffix against stored keys
- paths with backslashes: normalized prior to matching
"""
all_metadata = _load_lang_metadata(lang_dir)
if not all_metadata:
return {}
normalized = _normalize_source_key(source_file)
# Direct hit (already relative POSIX-style)
if normalized in all_metadata:
return all_metadata.get(normalized, {})
# If absolute or otherwise not directly found, try suffix matching against stored relative keys
# Build candidate suffixes from the provided path
parts = normalized.split("/")
best_key = None
best_len = -1
for i in range(len(parts)):
suffix = "/".join(parts[i:])
if suffix in all_metadata and len(suffix) > best_len:
best_key = suffix
best_len = len(suffix)
if best_key is not None:
return all_metadata.get(best_key, {})
return {}
def remove_text_metadata_for_source(lang_dir: Path, source_file: str | Path) -> None:
normalized_key = _normalize_source_key(source_file)
lock = _get_lock_for_path(_get_metadata_file_path(lang_dir))
with lock:
all_metadata = _load_lang_metadata(lang_dir)
changed = False
if normalized_key in all_metadata:
del all_metadata[normalized_key]
changed = True
# Also try absolute key variant if source_file was a Path
try:
absolute_key = _normalize_source_key(Path(source_file).resolve())
if absolute_key in all_metadata:
del all_metadata[absolute_key]
changed = True
except Exception:
pass
if changed:
_save_lang_metadata(lang_dir, all_metadata)
@@ -196,18 +196,14 @@ class TestJupyterNotebookTranslator:
assert all(isinstance(line, str) for line in cell_source)
@patch("co_op_translator.core.llm.jupyter_notebook_translator.MarkdownTranslator")
@patch(
"co_op_translator.core.llm.jupyter_notebook_translator.add_notebook_metadata"
)
@pytest.mark.asyncio
async def test_translate_notebook_adds_metadata(
async def test_translate_notebook_does_not_embed_metadata(
self,
mock_add_metadata,
mock_markdown_translator_class,
temp_notebook_file,
tmp_path,
):
"""Test that notebook translation adds coopTranslator metadata."""
"""Notebook translation should NOT embed coopTranslator metadata (centralized JSON is used)."""
# Setup mock translator
mock_translator = AsyncMock()
mock_translator.translate_markdown = AsyncMock(
@@ -215,38 +211,16 @@ class TestJupyterNotebookTranslator:
)
mock_markdown_translator_class.create.return_value = mock_translator
# Setup mock metadata function
expected_metadata = {
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3",
},
"coopTranslator": {
"original_hash": "test_hash",
"translation_date": "2025-01-26T14:30:00+00:00",
"source_file": "test.ipynb",
"language_code": "ko",
},
}
}
mock_add_metadata.return_value = {"cells": [], **expected_metadata}
# Create translator with root directory
# Create translator with root directory and translate without disclaimer for simplicity
translator = JupyterNotebookTranslator.create(tmp_path)
result = await translator.translate_notebook(temp_notebook_file, "ko")
result = await translator.translate_notebook(
temp_notebook_file, "ko", add_disclaimer=False
)
# Verify add_notebook_metadata was called with correct parameters
mock_add_metadata.assert_called_once()
call_args = mock_add_metadata.call_args
assert call_args[0][1] == temp_notebook_file # original_path
assert call_args[0][2] == "ko" # language_code
assert call_args[0][3] == tmp_path # root_dir
# Verify result includes metadata
# Verify result includes notebook metadata but not coopTranslator section
translated_notebook = json.loads(result)
assert "metadata" in translated_notebook
assert "coopTranslator" not in translated_notebook["metadata"]
@patch("co_op_translator.core.llm.jupyter_notebook_translator.MarkdownTranslator")
@pytest.mark.asyncio