From 5d44c6a3b4c588b0383ebfd9cfb0ba66ffc47327 Mon Sep 17 00:00:00 2001 From: Minseok Song <99078115+skytin1004@users.noreply.github.com> Date: Tue, 27 Jan 2026 18:21:56 +0900 Subject: [PATCH] Core: Remove inline file-top metadata and migrate to centralized JSON (#349) --- src/co_op_translator/cli/evaluate.py | 22 ++- .../core/llm/jupyter_notebook_translator.py | 13 +- .../core/llm/markdown_evaluator.py | 91 +++++++++---- .../core/project/directory_manager.py | 109 +++++++++++---- .../core/project/project_evaluator.py | 78 ++++++----- .../core/project/project_translator.py | 39 +++--- .../core/project/translation_manager.py | 114 +++++++++++++--- .../utils/common/metadata_utils.py | 128 ++++++++++++++++++ .../llm/test_jupyter_notebook_translator.py | 42 ++---- 9 files changed, 450 insertions(+), 186 deletions(-) diff --git a/src/co_op_translator/cli/evaluate.py b/src/co_op_translator/cli/evaluate.py index b75df959..764e6435 100644 --- a/src/co_op_translator/cli/evaluate.py +++ b/src/co_op_translator/cli/evaluate.py @@ -182,15 +182,29 @@ def evaluate_command( except ValueError: display_path = str(file_path).replace("\\", "/") - # Read file to get issues for reference + # Read centralized metadata to get issues for reference try: - with open(file_path, "r", encoding="utf-8") as f: - content = f.read() from co_op_translator.utils.common.metadata_utils import ( + read_text_metadata_for_source, extract_metadata_from_content, ) - metadata = extract_metadata_from_content(content) + trans_path = Path(file_path) + lang_dir = root_path / "translations" / language_code + try: + rel = trans_path.resolve().relative_to(lang_dir) + orig_path = root_path / rel + except Exception: + orig_path = None + + metadata = {} + if orig_path is not None: + metadata = read_text_metadata_for_source(lang_dir, orig_path) + if not metadata: + with open(file_path, "r", encoding="utf-8") as f: + content = f.read() + metadata = extract_metadata_from_content(content) + issues = [] if metadata and "evaluation" in metadata: # Only check issues field diff --git a/src/co_op_translator/core/llm/jupyter_notebook_translator.py b/src/co_op_translator/core/llm/jupyter_notebook_translator.py index a75b24f8..c1728736 100644 --- a/src/co_op_translator/core/llm/jupyter_notebook_translator.py +++ b/src/co_op_translator/core/llm/jupyter_notebook_translator.py @@ -10,10 +10,7 @@ import logging from pathlib import Path from .markdown_translator import MarkdownTranslator -from co_op_translator.utils.common.metadata_utils import ( - calculate_file_hash, - add_notebook_metadata, -) +from co_op_translator.utils.common.metadata_utils import calculate_file_hash logger = logging.getLogger(__name__) @@ -164,14 +161,6 @@ class JupyterNotebookTranslator: notebook["cells"].append(disclaimer_cell) logger.debug(f"Added disclaimer cell to {notebook_path.name}") - # Add coopTranslator metadata to the notebook - notebook_path = Path(notebook_path) - notebook = add_notebook_metadata( - notebook, notebook_path, language_code, self.root_dir - ) - - logger.debug(f"Added coopTranslator metadata to {notebook_path.name}") - # Return the modified notebook as JSON string return json.dumps(notebook, ensure_ascii=False, indent=1) diff --git a/src/co_op_translator/core/llm/markdown_evaluator.py b/src/co_op_translator/core/llm/markdown_evaluator.py index e3f866ce..530a99b6 100644 --- a/src/co_op_translator/core/llm/markdown_evaluator.py +++ b/src/co_op_translator/core/llm/markdown_evaluator.py @@ -8,7 +8,8 @@ from co_op_translator.config.llm_config.provider import LLMProvider from co_op_translator.utils.common.metadata_utils import ( extract_metadata_from_content, extract_content_without_metadata, - format_metadata_comment, + read_text_metadata_for_source, + save_text_metadata_for_source, ) from co_op_translator.utils.llm.markdown_utils import ( generate_evaluation_prompt, @@ -254,11 +255,46 @@ class MarkdownEvaluator(ABC): with open(translated_file, "r", encoding="utf-8") as f: translated_content = f.read() - # Extract existing metadata from translated file - metadata = extract_metadata_from_content(translated_content) - if not metadata: - logger.warning(f"No metadata found in {translated_file}") - return {}, False + # Determine language directory and original file path + translated_file = Path(translated_file) + + # Ascend to find the language directory containing .co-op-translator.json + def _find_lang_dir(path: Path) -> Path | None: + cur = path.parent + while True: + meta = cur / ".co-op-translator.json" + if meta.exists(): + return cur + parent = cur.parent + if parent == cur: + return None + cur = parent + + lang_dir = _find_lang_dir(translated_file) + if lang_dir is None: + # Fallback: try to locate 'translations/' in the path + parts = translated_file.resolve().parts + try: + idx = parts.index("translations") + lang_dir = Path(*parts[: idx + 2]) + except ValueError: + lang_dir = translated_file.parent + + try: + rel = translated_file.resolve().relative_to(lang_dir) + original_file = (self.root_dir or Path.cwd()) / rel + except Exception: + # As last resort, attempt to extract from legacy inline metadata + metadata_legacy = extract_metadata_from_content(translated_content) + source_file = ( + metadata_legacy.get("source_file") if metadata_legacy else None + ) + if not source_file: + logger.warning( + f"Unable to resolve original path for {translated_file}" + ) + return {}, False + original_file = (self.root_dir or Path.cwd()) / source_file # Initialize evaluation results rule_based_result = {"confidence_score": 1.0, "issues_found": []} @@ -371,29 +407,26 @@ class MarkdownEvaluator(ABC): rule_based_result, chunk_evaluations ) - # Update metadata with evaluation - metadata["evaluation"] = evaluation_result - - # Format updated metadata - metadata_comment = format_metadata_comment(metadata) - - # Update the file with new metadata - # Extract the actual content without metadata - content_without_metadata = extract_content_without_metadata( - translated_content - ) - - # Create updated content with new metadata followed by original content - updated_content = metadata_comment - if content_without_metadata.strip(): - updated_content += content_without_metadata - - # Write back to file - with open(translated_file, "w", encoding="utf-8") as f: - f.write(updated_content) - - logger.info(f"Updated evaluation metadata for {translated_file}") - return evaluation_result, True + # Save evaluation into centralized language metadata + language_code = lang_dir.name if lang_dir else "" + extra_fields = {"evaluation": evaluation_result} + try: + save_text_metadata_for_source( + lang_dir, + original_file, + language_code, + root_dir=self.root_dir, + extra_fields=extra_fields, + ) + logger.info( + f"Updated evaluation metadata for {translated_file} in centralized JSON" + ) + return evaluation_result, True + except Exception as e: + logger.error( + f"Failed to save evaluation metadata for {translated_file}: {e}" + ) + return {}, False except Exception as e: logger.error(f"Error evaluating {translated_file}: {e}") diff --git a/src/co_op_translator/core/project/directory_manager.py b/src/co_op_translator/core/project/directory_manager.py index 71d6d2ab..e2892f56 100644 --- a/src/co_op_translator/core/project/directory_manager.py +++ b/src/co_op_translator/core/project/directory_manager.py @@ -10,6 +10,7 @@ from pathlib import PurePosixPath from co_op_translator.utils.common.metadata_utils import ( extract_metadata_from_content, remove_image_metadata, + remove_text_metadata_for_source, ) from co_op_translator.config.constants import ( SUPPORTED_MARKDOWN_EXTENSIONS, @@ -210,29 +211,54 @@ class DirectoryManager: continue logger.info(f"Processing translation file: {trans_file}") - # Read translation file and extract metadata using utility - content = trans_file.read_text(encoding="utf-8") - metadata = extract_metadata_from_content(content) - if not metadata: - logger.warning(f"No metadata found in: {trans_file}") - continue - source_file = metadata.get("source_file") - if not source_file: - logger.warning(f"No source_file in metadata: {trans_file}") - continue + original_file = None + # Prefer legacy inline metadata if present to resolve original path robustly + try: + content = trans_file.read_text(encoding="utf-8") + metadata = extract_metadata_from_content(content) + source_file = ( + metadata.get("source_file") if metadata else None + ) + if source_file: + # Normalize backslashes and construct a proper relative Path + rel_parts = ( + str(source_file).replace("\\", "/").split("/") + ) + rel_path = Path(*rel_parts) + original_file = self.root_dir / rel_path + except Exception: + # Ignore content read/parse issues; will fallback to relative mapping + pass - normalized_path = str(PurePosixPath(source_file)) - original_file = self.root_dir / normalized_path + if original_file is None: + # Fallback: compute original path by relative path from language dir + try: + rel = trans_file.relative_to(translation_dir) + original_file = self.root_dir / rel + except ValueError: + logger.warning( + f"Unable to determine source for: {trans_file}" + ) + continue logger.info(f"Checking original file: {original_file}") if not original_file.exists(): logger.info( f"Original file not found, deleting: {trans_file}" ) - trans_file.unlink() - removed_count += 1 - logger.info(f"Successfully deleted: {trans_file}") + try: + trans_file.unlink() + removed_count += 1 + logger.info(f"Successfully deleted: {trans_file}") + finally: + # Remove centralized metadata entry for this source + try: + remove_text_metadata_for_source( + translation_dir, original_file + ) + except Exception: + pass parent = trans_file.parent while parent != translation_dir: @@ -282,28 +308,51 @@ class DirectoryManager: if not nb_file.exists(): continue - with open(nb_file, "r", encoding="utf-8") as f: - nb_json = json.load(f) - - coop_meta = nb_json.get("metadata", {}).get( - "coopTranslator", {} - ) - source_file = coop_meta.get("source_file") - if not source_file: - logger.warning( - f"No source_file in notebook metadata: {nb_file}" + original_file = None + # Prefer legacy notebook inline metadata if present + try: + with open(nb_file, "r", encoding="utf-8") as f: + nb_json = json.load(f) + coop_meta = nb_json.get("metadata", {}).get( + "coopTranslator", {} ) - continue + source_file = coop_meta.get("source_file") + if source_file: + rel_parts = ( + str(source_file).replace("\\", "/").split("/") + ) + rel_path = Path(*rel_parts) + original_file = self.root_dir / rel_path + except Exception: + # Ignore JSON read/parse issues; will fallback to relative mapping + pass - normalized_path = str(PurePosixPath(source_file)) - original_file = self.root_dir / normalized_path + if original_file is None: + # Fallback: compute original notebook path by relative path from language dir + try: + rel = nb_file.relative_to(translation_dir) + original_file = self.root_dir / rel + except ValueError: + logger.warning( + f"Unable to determine source for notebook: {nb_file}" + ) + continue if not original_file.exists(): logger.info( f"Original notebook not found, deleting: {nb_file}" ) - nb_file.unlink() - removed_count += 1 + try: + nb_file.unlink() + removed_count += 1 + finally: + # Remove centralized metadata entry for this source + try: + remove_text_metadata_for_source( + translation_dir, original_file + ) + except Exception: + pass parent = nb_file.parent while parent != translation_dir: diff --git a/src/co_op_translator/core/project/project_evaluator.py b/src/co_op_translator/core/project/project_evaluator.py index 43b78cd1..d3fb1f56 100644 --- a/src/co_op_translator/core/project/project_evaluator.py +++ b/src/co_op_translator/core/project/project_evaluator.py @@ -9,7 +9,10 @@ from tqdm import tqdm from co_op_translator.core.llm.markdown_evaluator import MarkdownEvaluator from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS -from co_op_translator.utils.common.metadata_utils import extract_metadata_from_content +from co_op_translator.utils.common.metadata_utils import ( + extract_metadata_from_content, + read_text_metadata_for_source, +) logger = logging.getLogger(__name__) @@ -98,6 +101,9 @@ class ProjectEvaluator: bar_format="{l_bar}{bar}| {n_fmt}/{total_fmt} [{elapsed}<{remaining}, {rate_fmt}]", ) + # Precompute language directory for centralized metadata + lang_dir = self.translations_dir / language_code + # Perform rule-based evaluation for all files for orig_file, trans_file in translation_pairs: try: @@ -108,11 +114,10 @@ class ProjectEvaluator: with open(trans_file, "r", encoding="utf-8") as f: translated_content = f.read() - # Extract existing metadata - metadata = extract_metadata_from_content(translated_content) + # Prefer centralized metadata; fall back to inline + metadata = read_text_metadata_for_source(lang_dir, orig_file) if not metadata: - logger.warning(f"No metadata found in {trans_file}") - continue + metadata = extract_metadata_from_content(translated_content) # Store file info for later LLM evaluation file_key = str(trans_file) @@ -399,23 +404,24 @@ class ProjectEvaluator: translation_pairs = await self._get_translation_pairs(language_code) low_confidence_translations = [] - for _, trans_file in translation_pairs: + # Prefer centralized JSON metadata + lang_dir = self.translations_dir / language_code + + for orig_file, trans_file in translation_pairs: try: - with open(trans_file, "r", encoding="utf-8") as f: - content = f.read() - - from co_op_translator.utils.common.metadata_utils import ( - extract_metadata_from_content, - ) - - metadata = extract_metadata_from_content(content) + metadata = read_text_metadata_for_source(lang_dir, orig_file) + if not metadata: + # Fallback to inline + with open(trans_file, "r", encoding="utf-8") as f: + content = f.read() + metadata = extract_metadata_from_content(content) if metadata and "evaluation" in metadata: confidence = metadata["evaluation"].get("confidence_score", 1.0) if confidence < threshold: low_confidence_translations.append((trans_file, confidence)) except Exception as e: - logger.error(f"Error reading metadata from {trans_file}: {e}") + logger.error(f"Error reading evaluation metadata for {trans_file}: {e}") return low_confidence_translations @@ -434,16 +440,15 @@ class ProjectEvaluator: translation_pairs = await self._get_translation_pairs(language_code) translations_with_issues = [] - for _, trans_file in translation_pairs: + lang_dir = self.translations_dir / language_code + + for orig_file, trans_file in translation_pairs: try: - with open(trans_file, "r", encoding="utf-8") as f: - content = f.read() - - from co_op_translator.utils.common.metadata_utils import ( - extract_metadata_from_content, - ) - - metadata = extract_metadata_from_content(content) + metadata = read_text_metadata_for_source(lang_dir, orig_file) + if not metadata: + with open(trans_file, "r", encoding="utf-8") as f: + content = f.read() + metadata = extract_metadata_from_content(content) if metadata and "evaluation" in metadata: confidence = metadata["evaluation"].get("confidence_score", 1.0) @@ -453,7 +458,9 @@ class ProjectEvaluator: (trans_file, confidence, issues) ) except Exception as e: - logger.error(f"Error reading metadata from {trans_file}: {e}") + logger.error( + f"Error reading evaluation metadata from {trans_file}: {e}" + ) return translations_with_issues @@ -473,16 +480,15 @@ class ProjectEvaluator: translation_pairs = await self._get_translation_pairs(language_code) problematic_translations = [] - for _, trans_file in translation_pairs: + lang_dir = self.translations_dir / language_code + + for orig_file, trans_file in translation_pairs: try: - with open(trans_file, "r", encoding="utf-8") as f: - content = f.read() - - from co_op_translator.utils.common.metadata_utils import ( - extract_metadata_from_content, - ) - - metadata = extract_metadata_from_content(content) + metadata = read_text_metadata_for_source(lang_dir, orig_file) + if not metadata: + with open(trans_file, "r", encoding="utf-8") as f: + content = f.read() + metadata = extract_metadata_from_content(content) if metadata and "evaluation" in metadata: confidence = metadata["evaluation"].get("confidence_score", 1.0) @@ -493,6 +499,8 @@ class ProjectEvaluator: (trans_file, confidence, issues) ) except Exception as e: - logger.error(f"Error reading metadata from {trans_file}: {e}") + logger.error( + f"Error reading evaluation metadata from {trans_file}: {e}" + ) return problematic_translations diff --git a/src/co_op_translator/core/project/project_translator.py b/src/co_op_translator/core/project/project_translator.py index 4bb23b90..73f3e8e0 100644 --- a/src/co_op_translator/core/project/project_translator.py +++ b/src/co_op_translator/core/project/project_translator.py @@ -266,33 +266,30 @@ class ProjectTranslator: for trans_file_path, confidence in low_confidence_files: trans_file = Path(trans_file_path) - # Extract metadata to get original file path + # Compute original file path by relative path from translations// try: - with open(trans_file, "r", encoding="utf-8") as f: - content = f.read() + rel = trans_file.resolve().relative_to(self.translations_dir) + # drop first part (lang code) + if len(rel.parts) < 2: + raise ValueError("Unexpected translation path structure") + lang_code = rel.parts[0] + orig_rel = Path(*rel.parts[1:]) + orig_file = (self.root_dir / orig_rel).resolve() - from co_op_translator.utils.common.metadata_utils import ( - extract_metadata_from_content, - ) - - metadata = extract_metadata_from_content(content) - - if metadata and "source_file" in metadata: - # Get original file path from metadata - orig_file = self.root_dir / metadata["source_file"] - - if orig_file.exists(): - files_to_retranslate.append((orig_file, confidence)) - else: - error_msg = f"Original source file not found for translation '{trans_file.name}': Expected file '{orig_file}' does not exist. The source may have been moved or deleted." - logger.warning(error_msg) - errors.append(error_msg) + if orig_file.exists(): + files_to_retranslate.append((orig_file, confidence)) else: - error_msg = f"Invalid translation metadata in file '{trans_file.name}': Missing source_file information. The file may be corrupted or not generated by Co-op Translator." + error_msg = ( + f"Original source file not found for translation '{trans_file.name}': " + f"Expected file '{orig_file}' does not exist. The source may have been moved or deleted." + ) logger.warning(error_msg) errors.append(error_msg) except Exception as e: - error_msg = f"Failed to read translation file '{trans_file.name}': {str(e)}. Check file permissions and encoding." + error_msg = ( + f"Failed to resolve original file for translation '{trans_file.name}': {str(e)}. " + f"Ensure translations directory structure matches translations//..." + ) logger.error(error_msg) errors.append(error_msg) diff --git a/src/co_op_translator/core/project/translation_manager.py b/src/co_op_translator/core/project/translation_manager.py index f3269588..5a2b8fd7 100644 --- a/src/co_op_translator/core/project/translation_manager.py +++ b/src/co_op_translator/core/project/translation_manager.py @@ -23,6 +23,10 @@ from co_op_translator.utils.common.metadata_utils import ( is_image_up_to_date, remove_image_metadata, cleanup_orphan_image_metadata, + save_text_metadata_for_source, + read_text_metadata_for_source, + extract_metadata_from_content, + extract_content_without_metadata, ) from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS from co_op_translator.core.llm.markdown_translator import MarkdownTranslator @@ -166,12 +170,13 @@ class TranslationManager: handle_empty_document(file_path, output_file) return str(output_file) - # Perform initial translation attempt + # Perform initial translation attempt (do not embed inline metadata; use centralized JSON instead) translated_content = await self.markdown_translator.translate_markdown( document, language_code, file_path, translation_types=self.translation_types, + add_metadata=False, add_disclaimer=self.add_disclaimer, ) if not translated_content: @@ -191,6 +196,7 @@ class TranslationManager: language_code, file_path, translation_types=self.translation_types, + add_metadata=False, add_disclaimer=self.add_disclaimer, ) if not translated_content: @@ -211,6 +217,14 @@ class TranslationManager: logger.info( f"Translated {file_path} to {language_code} and saved to {translated_path}" ) + # Save centralized text metadata for this source file in the language directory + lang_dir = self.translations_dir / language_code + save_text_metadata_for_source( + lang_dir, + file_path, + language_code, + root_dir=self.root_dir, + ) return str(translated_path) except Exception as e: logger.error(f"Failed to write translation to {translated_path}: {e}") @@ -265,6 +279,14 @@ class TranslationManager: logger.info( f"Translated {file_path} to {language_code} and saved to {translated_path}" ) + # Save centralized text metadata for this source notebook in the language directory + lang_dir = self.translations_dir / language_code + save_text_metadata_for_source( + lang_dir, + file_path, + language_code, + root_dir=self.root_dir, + ) return str(translated_path) except Exception as e: logger.error(f"Failed to write translation to {translated_path}: {e}") @@ -1186,34 +1208,84 @@ class TranslationManager: return True try: - # Handle notebook files differently from markdown files + # Handle notebook files using dedicated helper (centralized JSON preferred inside helper) if translation_file.suffix.lower() in self.supported_notebook_extensions: - # Use the dedicated notebook metadata comparison function return not is_notebook_up_to_date(original_file, translation_file) - # Handle markdown files with HTML comment metadata - content = translation_file.read_text(encoding="utf-8") - metadata_match = re.search( - r"", - content, - re.DOTALL, - ) - if not metadata_match: - return True - + # Determine language directory from translation path + lang_dir = None try: - metadata = json.loads(metadata_match.group(1)) - except json.JSONDecodeError: + rel = translation_file.resolve().relative_to(self.translations_dir) + lang_code = rel.parts[0] + lang_dir = self.translations_dir / lang_code + except Exception: + # Fallback: use the parent directory (may be incorrect for deeply nested paths) + lang_dir = translation_file.parent + + # Prefer centralized JSON metadata + metadata = read_text_metadata_for_source(lang_dir, original_file) + if metadata and isinstance(metadata, dict): + stored_hash = metadata.get("original_hash") + if stored_hash: + current_hash = calculate_file_hash(original_file) + return stored_hash != current_hash + + # Legacy fallback: read inline HTML comment metadata and migrate + try: + content = translation_file.read_text(encoding="utf-8") + except Exception: return True - # Determine if content has changed since last translation - original_hash = calculate_file_hash(original_file) - stored_hash = metadata.get("original_hash") + legacy_meta = extract_metadata_from_content(content) + stored_hash = ( + legacy_meta.get("original_hash") + if isinstance(legacy_meta, dict) + else None + ) + if stored_hash: + # Migrate legacy metadata into centralized JSON + extra_fields = {} + # Preserve original_hash and translation_date from legacy metadata if present + extra_fields["original_hash"] = stored_hash + if "translation_date" in legacy_meta: + extra_fields["translation_date"] = legacy_meta.get( + "translation_date" + ) - if not stored_hash: - return True + # Compute language code again for save (if available) + language_code = None + try: + if "rel" not in locals(): + rel = translation_file.resolve().relative_to( + self.translations_dir + ) + language_code = rel.parts[0] + except Exception: + language_code = None - return stored_hash != original_hash + if language_code: + save_text_metadata_for_source( + lang_dir, + original_file, + language_code, + root_dir=self.root_dir, + extra_fields=extra_fields, + ) + + # Remove inline metadata from the translated markdown file + cleaned = extract_content_without_metadata(content) + if cleaned != content: + try: + translation_file.write_text(cleaned, encoding="utf-8") + except Exception: + # Non-fatal; continue + pass + + current_hash = calculate_file_hash(original_file) + return stored_hash != current_hash + + # No metadata available; consider it outdated + return True except Exception: return True diff --git a/src/co_op_translator/utils/common/metadata_utils.py b/src/co_op_translator/utils/common/metadata_utils.py index affc4fe4..dfc77523 100644 --- a/src/co_op_translator/utils/common/metadata_utils.py +++ b/src/co_op_translator/utils/common/metadata_utils.py @@ -266,7 +266,20 @@ def is_notebook_up_to_date(original_path: Path, translated_path: Path) -> bool: Returns: True if translated notebook is up to date, False otherwise """ + try: + translated_path = Path(translated_path) + if not translated_path.exists(): + return False + + lang_dir = _find_lang_dir_for_translated_file(translated_path) + if lang_dir is not None: + metadata = read_text_metadata_for_source(lang_dir, original_path) + stored_hash = metadata.get("original_hash") + if stored_hash: + current_hash = calculate_file_hash(Path(original_path)) + return stored_hash == current_hash + stored_hash = read_notebook_metadata(translated_path, "coopTranslator").get( "original_hash" ) @@ -309,6 +322,23 @@ def add_notebook_metadata( return notebook +def _find_lang_dir_for_translated_file(translated_path: Path) -> Optional[Path]: + translated_path = Path(translated_path) + current = translated_path.parent + + while True: + metadata_file = _get_metadata_file_path(current) + if metadata_file.exists(): + return current + + parent = current.parent + if parent == current: + break + current = parent + + return None + + # ============================================================================ # Image-specific metadata utilities # ============================================================================ @@ -543,3 +573,101 @@ def is_image_up_to_date( except Exception as e: logger.debug(f"Error checking image up-to-date status: {e}") return False + + +def _normalize_source_key(source: str | Path) -> str: + return str(source).replace("\\", "/") + + +def load_language_metadata(lang_dir: Path) -> dict: + return _load_lang_metadata(lang_dir) + + +def save_language_metadata(lang_dir: Path, metadata: dict) -> None: + lock = _get_lock_for_path(_get_metadata_file_path(lang_dir)) + with lock: + existing = _load_lang_metadata(lang_dir) + existing.update(metadata) + _save_lang_metadata(lang_dir, existing) + + +def save_text_metadata_for_source( + lang_dir: Path, + original_file: Path, + language_code: str, + root_dir: Path | None = None, + extra_fields: Optional[dict] = None, +) -> dict: + metadata = create_metadata(original_file, language_code, root_dir) + if extra_fields: + metadata.update(extra_fields) + + source_key = metadata.get("source_file") + if not source_key: + return metadata + + normalized_key = _normalize_source_key(source_key) # relative, POSIX-style + + lock = _get_lock_for_path(_get_metadata_file_path(lang_dir)) + with lock: + all_metadata = _load_lang_metadata(lang_dir) + all_metadata[normalized_key] = metadata + _save_lang_metadata(lang_dir, all_metadata) + + return metadata + + +def read_text_metadata_for_source(lang_dir: Path, source_file: str | Path) -> dict: + """Read text metadata by source path. + + Keys are stored as repo-relative, POSIX-style paths. This lookup accepts any + of the following and resolves to the best matching relative key: + - relative POSIX-style path (exact match) + - absolute path (Windows or POSIX): matched by the longest suffix against stored keys + - paths with backslashes: normalized prior to matching + """ + all_metadata = _load_lang_metadata(lang_dir) + if not all_metadata: + return {} + + normalized = _normalize_source_key(source_file) + # Direct hit (already relative POSIX-style) + if normalized in all_metadata: + return all_metadata.get(normalized, {}) + + # If absolute or otherwise not directly found, try suffix matching against stored relative keys + # Build candidate suffixes from the provided path + parts = normalized.split("/") + best_key = None + best_len = -1 + for i in range(len(parts)): + suffix = "/".join(parts[i:]) + if suffix in all_metadata and len(suffix) > best_len: + best_key = suffix + best_len = len(suffix) + + if best_key is not None: + return all_metadata.get(best_key, {}) + + return {} + + +def remove_text_metadata_for_source(lang_dir: Path, source_file: str | Path) -> None: + normalized_key = _normalize_source_key(source_file) + lock = _get_lock_for_path(_get_metadata_file_path(lang_dir)) + with lock: + all_metadata = _load_lang_metadata(lang_dir) + changed = False + if normalized_key in all_metadata: + del all_metadata[normalized_key] + changed = True + # Also try absolute key variant if source_file was a Path + try: + absolute_key = _normalize_source_key(Path(source_file).resolve()) + if absolute_key in all_metadata: + del all_metadata[absolute_key] + changed = True + except Exception: + pass + if changed: + _save_lang_metadata(lang_dir, all_metadata) diff --git a/tests/co_op_translator/core/llm/test_jupyter_notebook_translator.py b/tests/co_op_translator/core/llm/test_jupyter_notebook_translator.py index be836540..729d6dfa 100644 --- a/tests/co_op_translator/core/llm/test_jupyter_notebook_translator.py +++ b/tests/co_op_translator/core/llm/test_jupyter_notebook_translator.py @@ -196,18 +196,14 @@ class TestJupyterNotebookTranslator: assert all(isinstance(line, str) for line in cell_source) @patch("co_op_translator.core.llm.jupyter_notebook_translator.MarkdownTranslator") - @patch( - "co_op_translator.core.llm.jupyter_notebook_translator.add_notebook_metadata" - ) @pytest.mark.asyncio - async def test_translate_notebook_adds_metadata( + async def test_translate_notebook_does_not_embed_metadata( self, - mock_add_metadata, mock_markdown_translator_class, temp_notebook_file, tmp_path, ): - """Test that notebook translation adds coopTranslator metadata.""" + """Notebook translation should NOT embed coopTranslator metadata (centralized JSON is used).""" # Setup mock translator mock_translator = AsyncMock() mock_translator.translate_markdown = AsyncMock( @@ -215,38 +211,16 @@ class TestJupyterNotebookTranslator: ) mock_markdown_translator_class.create.return_value = mock_translator - # Setup mock metadata function - expected_metadata = { - "metadata": { - "kernelspec": { - "display_name": "Python 3", - "language": "python", - "name": "python3", - }, - "coopTranslator": { - "original_hash": "test_hash", - "translation_date": "2025-01-26T14:30:00+00:00", - "source_file": "test.ipynb", - "language_code": "ko", - }, - } - } - mock_add_metadata.return_value = {"cells": [], **expected_metadata} - - # Create translator with root directory + # Create translator with root directory and translate without disclaimer for simplicity translator = JupyterNotebookTranslator.create(tmp_path) - result = await translator.translate_notebook(temp_notebook_file, "ko") + result = await translator.translate_notebook( + temp_notebook_file, "ko", add_disclaimer=False + ) - # Verify add_notebook_metadata was called with correct parameters - mock_add_metadata.assert_called_once() - call_args = mock_add_metadata.call_args - assert call_args[0][1] == temp_notebook_file # original_path - assert call_args[0][2] == "ko" # language_code - assert call_args[0][3] == tmp_path # root_dir - - # Verify result includes metadata + # Verify result includes notebook metadata but not coopTranslator section translated_notebook = json.loads(result) assert "metadata" in translated_notebook + assert "coopTranslator" not in translated_notebook["metadata"] @patch("co_op_translator.core.llm.jupyter_notebook_translator.MarkdownTranslator") @pytest.mark.asyncio