Core: Promote programmatic translation API (#394)

Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
This commit is contained in:
Minseok Song
2026-04-16 14:02:07 +09:00
committed by GitHub
co-authored by Copilot
parent 1355c2ccc1
commit 78e0082a2b
25 changed files with 3347 additions and 230 deletions
+5
View File
@@ -0,0 +1,5 @@
"""Public programmatic API for Co-op Translator."""
from co_op_translator.api.translation import run_translation
__all__ = ["run_translation"]
+528
View File
@@ -0,0 +1,528 @@
import importlib.resources
import logging
import os
from contextlib import contextmanager
from pathlib import Path
from typing import Iterable
import click
import yaml
from co_op_translator.config.base_config import Config
from co_op_translator.config.llm_config.config import LLMConfig
from co_op_translator.config.vision_config.config import VisionConfig
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
from co_op_translator.core.project.project_translator import ProjectTranslator
from co_op_translator.utils.common.file_utils import (
render_updated_readme_languages_table,
render_updated_readme_other_courses,
update_readme_languages_table,
update_readme_other_courses,
)
from co_op_translator.utils.common.lang_utils import normalize_language_codes
from co_op_translator.utils.common.logging_utils import setup_logging
from co_op_translator.utils.common.metadata_utils import (
normalize_language_codes_in_lang_metadata,
)
from co_op_translator.utils.common.token_estimation import estimate_translation_tokens
from co_op_translator.utils.common.word_estimation import estimate_translation_words
logger = logging.getLogger(__name__)
def compute_pretranslation_virtual_inputs(
root_path: Path,
translation_types: list[str],
repo_url: str | None = None,
) -> dict[Path, str]:
"""Return virtual source content for deterministic pre-translation rewrites."""
if "markdown" not in translation_types:
return {}
readme_path = (root_path / "README.md").resolve()
if not readme_path.exists():
return {}
original = readme_path.read_text(encoding="utf-8")
updated = render_updated_readme_languages_table(original, repo_url=repo_url)
updated = render_updated_readme_other_courses(updated)
if updated == original:
return {}
return {readme_path: updated}
def run_translation(
language_codes: str,
root_dir: str = ".",
update: bool = False,
images: bool = False,
markdown: bool = False,
notebook: bool = False,
debug: bool = False,
save_logs: bool = False,
yes: bool = True,
add_disclaimer: bool = False,
translations_dir: str | None = None,
image_dir: str | None = None,
root_dirs: Iterable[str] | None = None,
groups: Iterable[tuple[str, str | None]] | None = None,
repo_url: str | None = None,
dry_run: bool = False,
) -> None:
"""Programmatic translation entrypoint mirroring the translate CLI options."""
def _split_lang_placeholder(path: str) -> tuple[str, str | None]:
placeholder = "<lang>"
if placeholder not in path:
return path, None
prefix, suffix = path.split(placeholder, 1)
prefix = prefix.rstrip("/\\")
suffix = suffix.lstrip("/\\")
return prefix, (suffix or None)
def _run_single_group(
*,
language_codes: str,
root_dir: str,
update: bool,
images: bool,
markdown: bool,
notebook: bool,
debug: bool,
save_logs: bool,
yes: bool,
add_disclaimer: bool,
translations_dir: str | None,
image_dir: str | None,
lang_subdir: str | None,
repo_url: str | None,
dry_run: bool,
) -> None:
Config.check_configuration()
translation_types: list[str] = []
if markdown:
translation_types.append("markdown")
if images:
translation_types.append("images")
if notebook:
translation_types.append("notebook")
if not translation_types:
translation_types = ["markdown", "notebook", "images"]
if "images" in translation_types:
cv_available = VisionConfig.check_configuration()
if not cv_available:
raise RuntimeError(
"Image translation is enabled but Azure AI Service is not configured.\n"
"Please add AZURE_AI_SERVICE_API_KEY to your environment variables or use "
"translation_types without 'images'.\n"
"See the .env.template file for required variables."
)
click.echo(f"🚀 Translation mode: {', '.join(translation_types)}")
root_path = Path(root_dir).resolve()
if not root_path.exists():
raise ValueError(f"Root directory does not exist: {root_dir}")
if not root_path.is_dir():
raise ValueError(f"Root path is not a directory: {root_dir}")
log_file_path = setup_logging(
root_path, debug=debug, save_logs=save_logs, command_name="translate"
)
if debug:
logging.debug("Debug mode enabled.")
if save_logs and log_file_path is not None:
click.echo(f"📄 Logs will be saved to: {log_file_path}")
LLMConfig.validate_connectivity()
logger.info("LLM health check passed.")
click.echo("✅ LLM health check passed.")
if "images" in translation_types:
VisionConfig.validate_connectivity()
logger.info("Vision health check passed.")
click.echo("✅ Vision health check passed.")
all_languages_selected = language_codes == "all"
if all_languages_selected:
click.echo(
"Warning: Translating all languages at once can take a significant amount of time, "
"especially for large projects."
)
if yes:
logger.info("Auto-confirming 'all' languages in non-interactive mode.")
click.echo("Auto-confirming translation for all languages...")
try:
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file)
if not font_mappings:
raise RuntimeError("Empty font mappings file")
language_codes = " ".join(
[
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
)
if not language_codes:
raise RuntimeError(
"No valid language codes found in font mappings"
)
logging.debug(
f"Loaded language codes from font mapping: {language_codes}"
)
except (FileNotFoundError, yaml.YAMLError) as e:
raise RuntimeError(f"Failed to load font mappings: {str(e)}") from e
if all_languages_selected:
try:
lang_list = Config.get_language_codes()
except Exception:
lang_list = [
code.strip() for code in language_codes.split() if code.strip()
]
else:
lang_list = [
code.strip() for code in language_codes.split() if code.strip()
]
lang_list = normalize_language_codes(lang_list) if lang_list else []
if update:
click.echo(
f"Warning: Update mode will delete all existing translations for '{language_codes}' "
f"and re-translate them."
)
try:
effective_translations_dir = (
(root_path / translations_dir).resolve()
if translations_dir is not None
and not Path(translations_dir).is_absolute()
else (
Path(translations_dir).resolve()
if translations_dir is not None
else (root_path / "translations")
)
)
effective_image_dir = (
(root_path / image_dir).resolve()
if image_dir is not None and not Path(image_dir).is_absolute()
else (
Path(image_dir).resolve()
if image_dir is not None
else (root_path / "translated_images")
)
)
migrator = LanguageFolderMigrator(
root_path,
translations_dir=effective_translations_dir,
image_dir=effective_image_dir,
)
alias_entries = migrator.detect_alias_folders()
if alias_entries:
canon_set = set(lang_list)
relevant = [e for e in alias_entries if e.canonical in canon_set]
if relevant:
plan = LanguageFolderMigrator.format_plan(relevant)
logger.info("Language folder migration plan:\n%s", plan)
click.echo(plan)
if dry_run:
click.echo("Dry run: no changes will be made.")
else:
renamed, msgs = migrator.execute(relevant, dry_run=False)
logger.info("Auto-migrated %d language folder(s).", renamed)
for m in msgs:
logger.warning(m)
except Exception as e: # pragma: no cover
logger.warning(f"Language folder migration step skipped: {e}")
if not dry_run:
try:
for lang in lang_list:
lang_root = (
effective_translations_dir
if "effective_translations_dir" in locals()
else (root_path / "translations")
) / lang
if lang_subdir:
lang_root = lang_root / lang_subdir
normalize_language_codes_in_lang_metadata(
lang_root,
lang,
)
normalize_language_codes_in_lang_metadata(
(
effective_image_dir
if "effective_image_dir" in locals()
else (root_path / "translated_images")
)
/ lang,
lang,
)
normalize_language_codes_in_lang_metadata(
root_path / "translated_images_fast" / lang,
lang,
)
except Exception as e: # pragma: no cover
logger.debug(f"Metadata normalization skipped: {e}")
readme_path = root_path / "README.md"
try:
if update_readme_languages_table(readme_path, repo_url=repo_url):
click.echo("✅ Updated README languages table from template.")
else:
click.echo(
"ℹ️ README languages table not updated (markers missing or template unavailable)."
)
except Exception as e: # pragma: no cover
logger.warning(f"Failed to update README languages table: {e}")
try:
if update_readme_other_courses(readme_path):
click.echo("✅ Updated README 'Other courses' section from template.")
except Exception as e: # pragma: no cover
logger.warning(f"Failed to update README 'Other courses': {e}")
translator = ProjectTranslator(
language_codes,
root_dir,
translation_types=translation_types,
add_disclaimer=add_disclaimer,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
if dry_run:
click.echo("🧪 Dry run complete: no changes made.")
return
translator.translate_project(
update=update,
)
logger.info(f"Project translation completed for languages: {language_codes}")
def _merge_estimates(
current: dict[str, int],
incoming: dict[str, int],
) -> dict[str, int]:
merged = dict(current)
for key in (
"markdown",
"notebook",
"images",
"outdated_markdown",
"outdated_notebook",
"outdated_images",
"outdated",
"total",
"words",
):
merged[key] = int(merged.get(key, 0)) + int(incoming.get(key, 0))
return merged
def _echo_estimate_summary(
est: dict[str, int],
translation_types: list[str],
) -> None:
translation_parts: list[str] = []
if "markdown" in translation_types:
translation_parts.append(f"markdown: {est.get('markdown', 0):,}")
if "notebook" in translation_types:
translation_parts.append(f"notebook: {est.get('notebook', 0):,}")
if "images" in translation_types:
translation_parts.append(f"images: {est.get('images', 0):,}")
retranslation_parts: list[str] = []
if "markdown" in translation_types:
retranslation_parts.append(
f"outdated markdowns: {est.get('outdated_markdown', 0):,}"
)
if "notebook" in translation_types:
retranslation_parts.append(
f"outdated notebooks: {est.get('outdated_notebook', 0):,}"
)
if "images" in translation_types:
retranslation_parts.append(
f"outdated images: {est.get('outdated_images', 0):,}"
)
breakdown_sections: list[str] = []
if translation_parts:
breakdown_sections.append(f"translation: {'; '.join(translation_parts)}")
if retranslation_parts:
breakdown_sections.append(
f"retranslation: {'; '.join(retranslation_parts)}"
)
breakdown = " | ".join(breakdown_sections) if breakdown_sections else "none"
click.echo(
"📊 Estimated translation volume before translation: "
f"{est.get('total', 0):,} tokens ({est.get('words', 0):,} words) "
f"(breakdown: {breakdown})"
)
def _compute_estimate_for_group(
*,
language_codes: str,
root_dir: str,
update: bool,
markdown: bool,
images: bool,
notebook: bool,
add_disclaimer: bool,
translations_dir: str | None,
image_dir: str | None,
lang_subdir: str | None,
repo_url: str | None,
) -> dict[str, int]:
translation_types: list[str] = []
if markdown:
translation_types.append("markdown")
if images:
translation_types.append("images")
if notebook:
translation_types.append("notebook")
if not translation_types:
translation_types = ["markdown", "notebook", "images"]
translator = ProjectTranslator(
language_codes,
root_dir,
translation_types=translation_types,
add_disclaimer=add_disclaimer,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
virtual_file_contents = compute_pretranslation_virtual_inputs(
Path(root_dir).resolve(),
translation_types,
repo_url=repo_url,
)
est = estimate_translation_tokens(
translator.translation_manager,
update=update,
virtual_file_contents=virtual_file_contents,
)
words_est = estimate_translation_words(
translator.translation_manager,
update=update,
virtual_file_contents=virtual_file_contents,
)
return {
"markdown": int(est.get("markdown", 0) or 0),
"notebook": int(est.get("notebook", 0) or 0),
"images": int(est.get("images", 0) or 0),
"outdated_markdown": int(est.get("outdated_markdown", 0) or 0),
"outdated_notebook": int(est.get("outdated_notebook", 0) or 0),
"outdated_images": int(est.get("outdated_images", 0) or 0),
"outdated": int(est.get("outdated", 0) or 0),
"total": int(est.get("total", 0) or 0),
"words": int(words_est.get("total", 0) or 0),
}
@contextmanager
def _tqdm_disabled(disabled: bool):
if not disabled:
yield
return
previous = os.environ.get("TQDM_DISABLE")
os.environ["TQDM_DISABLE"] = "1"
try:
yield
finally:
if previous is None:
os.environ.pop("TQDM_DISABLE", None)
else:
os.environ["TQDM_DISABLE"] = previous
aggregate_template = {
"markdown": 0,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 0,
"words": 0,
}
translation_types_for_summary: list[str] = []
if markdown:
translation_types_for_summary.append("markdown")
if images:
translation_types_for_summary.append("images")
if notebook:
translation_types_for_summary.append("notebook")
if not translation_types_for_summary:
translation_types_for_summary = ["markdown", "notebook", "images"]
execution_targets: list[tuple[str, str | None, str | None]] = []
if groups is not None:
for per_root, per_translations in list(groups):
per_translations_dir: str | None = per_translations
per_lang_subdir: str | None = None
if per_translations is not None:
base_part, suffix = _split_lang_placeholder(per_translations)
per_translations_dir = base_part or None
per_lang_subdir = suffix
execution_targets.append((per_root, per_translations_dir, per_lang_subdir))
elif root_dirs is not None:
for per_root in list(root_dirs):
execution_targets.append((per_root, translations_dir, None))
else:
execution_targets.append((root_dir, translations_dir, None))
aggregated_estimate = dict(aggregate_template)
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
group_estimate = _compute_estimate_for_group(
language_codes=language_codes,
root_dir=per_root,
update=update,
markdown=markdown,
images=images,
notebook=notebook,
add_disclaimer=add_disclaimer,
translations_dir=per_translations_dir,
image_dir=image_dir,
lang_subdir=per_lang_subdir,
repo_url=repo_url,
)
aggregated_estimate = _merge_estimates(aggregated_estimate, group_estimate)
_echo_estimate_summary(aggregated_estimate, translation_types_for_summary)
multi_group_mode = len(execution_targets) > 1
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
with _tqdm_disabled(multi_group_mode):
_run_single_group(
language_codes=language_codes,
root_dir=per_root,
update=update,
images=images,
markdown=markdown,
notebook=notebook,
debug=debug,
save_logs=save_logs,
yes=yes,
add_disclaimer=add_disclaimer,
translations_dir=per_translations_dir,
image_dir=image_dir,
lang_subdir=per_lang_subdir,
repo_url=repo_url,
dry_run=dry_run,
)
+46
View File
@@ -353,6 +353,52 @@ def translate_command(
add_disclaimer=add_disclaimer,
)
# Estimate tokens before running translation and print a concise summary
try:
est = translator.translation_manager.estimate_tokens(update=update)
translation_parts = []
if "markdown" in translation_types:
translation_parts.append(f"markdown: {est.get('markdown', 0):,}")
if "notebook" in translation_types:
translation_parts.append(f"notebook: {est.get('notebook', 0):,}")
if "images" in translation_types:
translation_parts.append(f"images: {est.get('images', 0):,}")
retranslation_parts = []
if "markdown" in translation_types:
retranslation_parts.append(
f"outdated markdowns: {est.get('outdated_markdown', 0):,}"
)
if "notebook" in translation_types:
retranslation_parts.append(
f"outdated notebooks: {est.get('outdated_notebook', 0):,}"
)
if "images" in translation_types:
retranslation_parts.append(
f"outdated images: {est.get('outdated_images', 0):,}"
)
breakdown_sections = []
if translation_parts:
breakdown_sections.append(
f"translation: {'; '.join(translation_parts)}"
)
if retranslation_parts:
breakdown_sections.append(
f"retranslation: {'; '.join(retranslation_parts)}"
)
breakdown = " | ".join(breakdown_sections) if breakdown_sections else "none"
click.echo(
f"📊 Estimated tokens before translation: {est.get('total', 0):,} (breakdown: {breakdown})"
)
except Exception as e:
logger.debug(f"Failed to compute estimated tokens: {e}")
# If dry-run, stop after estimation without making any changes
if dry_run:
click.echo("🧪 Dry run complete: no changes made.")
return
# Update README shared sections BEFORE translation
readme_path = root_path / "README.md"
try:
@@ -0,0 +1,83 @@
# Frontmatter Translation Configuration
#
# This file defines which frontmatter fields should be preserved (not translated)
# and which should be translated when processing markdown files.
#
# Usage:
# - preserve: Fields that should NEVER be translated (e.g., slug, id, technical identifiers)
# - translate: Fields that should be translated (e.g., title, description, user-facing text)
#
# Note: This configuration provides deterministic control over frontmatter translation,
# ensuring that technical fields are never accidentally modified by the LLM.
frontmatter:
# Fields to preserve as-is (never translate)
preserve:
- slug
- id
- order
- section
- sidebar_position
- sidebar_label # Often used as a fallback, but sometimes contains translatable text
- date
- author
- tags # Array of technical tags
- categories # Array of technical categories
- draft
- published
- layout
- template
- type
- weight
- url
- permalink
- redirect_from
- redirect_to
- canonical_url
- robots
- sitemap
- priority
- changefreq
- lastmod
- image # Path to image file
- cover
- thumbnail
- featured_image
- og_image
- twitter_image
- icon
- emoji
- color
- badge
- status
- version
- api_version
- schema_version
# Fields to translate
translate:
- title
- description
- excerpt
- summary
- abstract
- subtitle
- tagline
- caption
- alt
- label # User-facing label text
- placeholder
- tooltip
- help_text
- error_message
- success_message
- warning_message
- info_message
- og_title
- og_description
- twitter_title
- twitter_description
- meta_title
- meta_description
- seo_title
- seo_description
@@ -27,6 +27,7 @@ class JupyterNotebookTranslator:
root_dir: Path = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
):
"""Initialize the notebook translator.
@@ -36,10 +37,12 @@ class JupyterNotebookTranslator:
self.root_dir = root_dir
self.translations_dir = translations_dir
self.image_dir = image_dir
self.lang_subdir = Path(lang_subdir) if lang_subdir else None
self.markdown_translator = MarkdownTranslator.create(
root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=self.lang_subdir,
)
async def translate_notebook(
@@ -170,6 +173,7 @@ class JupyterNotebookTranslator:
root_dir: Path = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
) -> "JupyterNotebookTranslator":
"""Create a Jupyter Notebook translator instance.
@@ -181,4 +185,9 @@ class JupyterNotebookTranslator:
Returns:
JupyterNotebookTranslator instance
"""
return cls(root_dir, translations_dir=translations_dir, image_dir=image_dir)
return cls(
root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
@@ -1,6 +1,7 @@
from abc import ABC, abstractmethod
import asyncio
import logging
import os
import re
from pathlib import Path
from importlib import resources
@@ -16,6 +17,10 @@ from co_op_translator.utils.llm.markdown_utils import (
normalize_internal_anchor_links,
SPLIT_DELIMITER,
)
from co_op_translator.utils.llm.frontmatter_utils import (
get_frontmatter_parser,
adjust_frontmatter_links,
)
from co_op_translator.utils.llm.code_comment_translator import (
translate_comments_in_code_blocks,
)
@@ -43,6 +48,7 @@ class MarkdownTranslator(ABC):
root_dir: Path | None = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
):
"""Initialize translator with project configuration.
@@ -52,6 +58,7 @@ class MarkdownTranslator(ABC):
self.root_dir = root_dir
self.translations_dir = translations_dir
self.image_dir = image_dir
self.lang_subdir = Path(lang_subdir) if lang_subdir else None
self.font_config = FontConfig()
def calculate_file_hash(self, file_path: Path) -> str:
@@ -91,6 +98,19 @@ class MarkdownTranslator(ABC):
"""
return format_metadata_comment(metadata)
@staticmethod
def _insert_metadata_comment(content: str, metadata_comment: str) -> str:
if not metadata_comment:
return content
frontmatter_pattern = r"^---[ \t]*\n.*?\n---[ \t]*\n"
match = re.match(frontmatter_pattern, content, re.DOTALL)
if match:
end = match.end()
return content[:end] + "\n" + metadata_comment + content[end:]
return metadata_comment + content
async def translate_markdown(
self,
document: str,
@@ -103,7 +123,8 @@ class MarkdownTranslator(ABC):
"""Translate markdown document to target language.
Handles complex documents by splitting into manageable chunks while
preserving formatting, links, and code blocks.
preserving formatting, links, and code blocks. Frontmatter fields are
handled deterministically based on configuration.
Args:
document: Content of the markdown file
@@ -132,11 +153,40 @@ class MarkdownTranslator(ABC):
language_name = self.font_config.get_language_name(language_code)
is_rtl = self.font_config.is_rtl(language_code)
# Step 0: Extract and process frontmatter
parser = get_frontmatter_parser()
frontmatter, body = parser.extract_frontmatter(document)
preserve_fields = {}
translate_fields = {}
frontmatter_section = ""
if frontmatter:
# Split frontmatter into preserve and translate fields
preserve_fields, translate_fields = parser.split_fields(frontmatter)
logger.debug(
f"Frontmatter split for '{md_file_path.name}': "
f"{len(preserve_fields)} preserve, {len(translate_fields)} translate"
)
# Convert translatable fields to markdown for LLM
if translate_fields:
frontmatter_section = parser.extract_translatable_fields_as_markdown(
translate_fields
)
logger.debug(
f"Translatable frontmatter fields for '{md_file_path.name}': "
f"{list(translate_fields.keys())}"
)
# Use body for translation (frontmatter already extracted)
document_to_translate = body
# Step 1: Replace code blocks and inline code with placeholders
(
document_with_placeholders,
placeholder_map,
) = replace_code_blocks(document)
) = replace_code_blocks(document_to_translate)
# Step 1.5: Translate only the comments inside fenced code blocks
placeholder_map = await translate_comments_in_code_blocks(
@@ -172,7 +222,64 @@ class MarkdownTranslator(ABC):
# Step 4.75: Restore the code blocks and inline code from placeholders
translated_content = restore_code_blocks(translated_content, placeholder_map)
# Step 5: Update links
# Step 5: Translate frontmatter fields if any
translated_frontmatter_fields = {}
if frontmatter_section:
# Translate the frontmatter section
frontmatter_prompt = generate_prompt_template(
language_code, language_name, frontmatter_section, is_rtl
)
try:
translated_fm_markdown = await asyncio.wait_for(
self._run_prompt(frontmatter_prompt, "frontmatter", 1),
timeout=self.TRANSLATION_TIMEOUT_SECONDS,
)
# Parse translated fields back from markdown
translated_frontmatter_fields = (
parser.parse_translated_fields_from_markdown(
translated_fm_markdown, translate_fields
)
)
logger.debug(
f"Translated frontmatter fields for '{md_file_path.name}': "
f"{list(translated_frontmatter_fields.keys())}"
)
except asyncio.TimeoutError:
logger.warning(
f"Frontmatter translation timeout for '{md_file_path.name}': "
f"Using original values for translatable fields."
)
translated_frontmatter_fields = translate_fields
except Exception as e:
logger.error(
f"Frontmatter translation failed for '{md_file_path.name}': {e}. "
f"Using original values for translatable fields."
)
translated_frontmatter_fields = translate_fields
# Step 6: Merge frontmatter and reconstruct
if frontmatter:
merged_frontmatter = parser.merge_fields(
preserve_fields, translated_frontmatter_fields
)
# Step 6.5: Adjust frontmatter links (same logic as markdown-only mode)
adjusted_frontmatter = adjust_frontmatter_links(
merged_frontmatter,
md_file_path,
language_code,
self.root_dir,
self.translations_dir,
self.image_dir,
translation_types,
lang_subdir=self.lang_subdir,
)
translated_content = parser.reconstruct_content(
adjusted_frontmatter, translated_content
)
# Step 7: Update links
updated_content = update_links(
md_file_path,
translated_content,
@@ -183,10 +290,10 @@ class MarkdownTranslator(ABC):
translation_types=translation_types,
)
# Step 6: Add metadata and disclaimer (only if requested)
# Step 8: Add metadata and disclaimer (only if requested)
result = updated_content
if add_metadata:
result = metadata_comment + result
result = self._insert_metadata_comment(updated_content, metadata_comment)
if add_disclaimer:
disclaimer = await self.generate_disclaimer(language_code)
if disclaimer:
@@ -320,6 +427,7 @@ class MarkdownTranslator(ABC):
root_dir: Path | None = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
) -> "MarkdownTranslator":
"""Create appropriate markdown translator based on configured provider.
@@ -350,6 +458,7 @@ class MarkdownTranslator(ABC):
root_dir=root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
elif provider == LLMProvider.OPENAI:
from co_op_translator.core.llm.providers.openai.markdown_translator import (
@@ -360,6 +469,7 @@ class MarkdownTranslator(ABC):
root_dir=root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
else:
raise ValueError(
@@ -25,6 +25,7 @@ class AzureMarkdownTranslator(MarkdownTranslator):
root_dir: Path | None = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
):
"""Initialize translator with Azure-specific configuration.
@@ -32,7 +33,10 @@ class AzureMarkdownTranslator(MarkdownTranslator):
root_dir: Optional root directory for the project
"""
super().__init__(
root_dir, translations_dir=translations_dir, image_dir=image_dir
root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
self.kernel = self._initialize_kernel()
active = AzureOpenAIConfig.get_active_env_set()
@@ -25,6 +25,7 @@ class OpenAIMarkdownTranslator(MarkdownTranslator):
root_dir: Path | None = None,
translations_dir: Path | None = None,
image_dir: Path | None = None,
lang_subdir: Path | None = None,
):
"""Initialize translator with OpenAI configuration.
@@ -32,7 +33,10 @@ class OpenAIMarkdownTranslator(MarkdownTranslator):
root_dir: Optional root directory for the project
"""
super().__init__(
root_dir, translations_dir=translations_dir, image_dir=image_dir
root_dir,
translations_dir=translations_dir,
image_dir=image_dir,
lang_subdir=lang_subdir,
)
self.kernel = self._initialize_kernel()
active = OpenAIConfig.get_active_env_set()
@@ -81,7 +81,11 @@ class TextTranslator(ABC):
messages=[
{
"role": "system",
"content": "You are a translator. Return exactly the same number of translations as input lines.",
"content": (
"You are a translator. Return exactly the same number of translations "
"as input lines. Treat any glossary or rules section as instructions only, "
"never as extra lines to translate."
),
},
{"role": "user", "content": prompt},
],
@@ -35,6 +35,7 @@ class DirectoryManager:
language_codes: list[str],
excluded_dirs: list[str],
image_dir: Path | None = None,
lang_subdir: Path | None = None,
):
"""Initialize directory manager with project configuration.
@@ -44,6 +45,7 @@ class DirectoryManager:
language_codes: List of target language codes
excluded_dirs: List of directories to exclude
image_dir: Directory for translated images (flat tree, language code embedded in filename)
lang_subdir: Optional nested subdirectory within each language folder
"""
self.root_dir = root_dir
self.translations_dir = translations_dir
@@ -55,6 +57,21 @@ class DirectoryManager:
if image_dir is not None
else (self.root_dir / "translated_images")
)
self.lang_subdir = Path(lang_subdir) if lang_subdir else None
def _get_language_root(self, language_code: str) -> Path:
"""Get the root directory for a specific language's translations.
Args:
language_code: The target language code (e.g., 'ko', 'fr')
Returns:
Path to the language-specific translation directory
"""
lang_dir = self.translations_dir / language_code
if self.lang_subdir:
lang_dir = lang_dir / self.lang_subdir
return lang_dir
def sync_directory_structure(
self, markdown: bool = True, images: bool = True, notebooks: bool = True
@@ -106,7 +123,7 @@ class DirectoryManager:
# Sync each language directory
for lang_code in self.language_codes:
lang_dir = self.translations_dir / lang_code
lang_dir = self._get_language_root(lang_code)
if not lang_dir.exists():
lang_dir.mkdir(parents=True)
logger.info(f"Created language directory: {lang_dir}")
@@ -190,7 +207,7 @@ class DirectoryManager:
# Handle markdown files
if markdown:
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
translation_dir = self._get_language_root(lang_code)
if not translation_dir.exists():
logger.info(
f"Translation directory does not exist: {translation_dir}"
@@ -288,7 +305,7 @@ class DirectoryManager:
# Handle notebook files
if notebooks:
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
translation_dir = self._get_language_root(lang_code)
if not translation_dir.exists():
logger.info(
f"Notebook translation directory does not exist: {translation_dir}"
@@ -436,7 +453,6 @@ class DirectoryManager:
rel_parts = image_file.relative_to(image_dir).parts
except Exception:
rel_parts = ()
lang_code = None
# Accept alias language folder names by normalizing to canonical
if len(rel_parts) >= 2:
@@ -454,6 +470,7 @@ class DirectoryManager:
path_hash_segment = parts[-3]
base_name = ".".join(parts[:-3])
# If language code is not supported (not in language_codes), delete it
if lang_code not in self.language_codes:
try:
image_file.unlink()
@@ -22,6 +22,8 @@ from co_op_translator.utils.common.lang_utils import (
from .directory_manager import DirectoryManager
from .translation_manager import TranslationManager
from co_op_translator.utils.common.file_utils import read_input_file
from co_op_translator.utils.common.token_estimation import count_tokens
logger = logging.getLogger(__name__)
@@ -41,6 +43,7 @@ class ProjectTranslator:
add_disclaimer: bool = True,
translations_dir=None,
image_dir=None,
lang_subdir=None,
):
"""Initialize project translation environment.
@@ -54,6 +57,7 @@ class ProjectTranslator:
# Normalize to canonical BCP 47 (accept alias input like tw/cn/br)
self.language_codes = normalize_language_codes(language_codes.split())
self.root_dir = Path(root_dir).resolve()
self.lang_subdir = Path(lang_subdir) if lang_subdir else None
# Resolve translations_dir relative to root_dir when a relative path is provided.
if translations_dir is not None:
t_dir = Path(translations_dir)
@@ -133,6 +137,7 @@ class ProjectTranslator:
self.root_dir,
translations_dir=self.translations_dir,
image_dir=self.image_dir,
lang_subdir=self.lang_subdir,
)
# Initialize notebook translator if notebooks are enabled
@@ -141,6 +146,7 @@ class ProjectTranslator:
self.root_dir,
translations_dir=self.translations_dir,
image_dir=self.image_dir,
lang_subdir=self.lang_subdir,
)
else:
self.notebook_translator = None
@@ -152,6 +158,7 @@ class ProjectTranslator:
self.language_codes,
self.excluded_dirs,
image_dir=self.image_dir,
lang_subdir=self.lang_subdir,
)
self.translation_manager = TranslationManager(
self.root_dir,
@@ -166,6 +173,7 @@ class ProjectTranslator:
self.notebook_translator,
self.translation_types,
add_disclaimer=add_disclaimer,
lang_subdir=self.lang_subdir,
)
def translate_project(
@@ -267,6 +275,7 @@ class ProjectTranslator:
excluded_dirs=self.excluded_dirs,
use_llm=True,
use_rule=True,
lang_subdir=self.lang_subdir,
)
# Get low confidence files
@@ -318,6 +327,21 @@ class ProjectTranslator:
logger.info("No files could be prepared for retranslation")
return 0, errors
# Estimate tokens for these retranslation sources (single language)
try:
est_total = 0
for orig_file, _ in files_to_retranslate:
try:
text = read_input_file(orig_file)
est_total += count_tokens(text)
except Exception:
continue
logger.info(
f"Estimated tokens for selected low-confidence retranslation targets: {est_total:,} (files: {len(files_to_retranslate)})"
)
except Exception as e:
logger.debug(f"Failed to estimate tokens for low-confidence set: {e}")
# Retranslate files
retranslated = 0
@@ -27,11 +27,13 @@ from co_op_translator.utils.common.metadata_utils import (
read_text_metadata_for_source,
extract_metadata_from_content,
extract_content_without_metadata,
normalize_language_codes_in_lang_metadata,
)
from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS
from co_op_translator.core.llm.markdown_translator import MarkdownTranslator
from co_op_translator.core.project.directory_manager import DirectoryManager
from co_op_translator.utils.common.task_utils import worker
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
from co_op_translator.utils.llm.markdown_utils import (
compare_line_breaks,
update_image_links,
@@ -41,6 +43,11 @@ from co_op_translator.config.base_config import Config
from co_op_translator.utils.common.file_utils import (
canonicalize_image_links_in_translations,
)
from co_op_translator.utils.common.token_estimation import (
estimate_tokens_for_outdated,
estimate_tokens_for_sources,
estimate_translation_tokens,
)
logger = logging.getLogger(__name__)
@@ -66,6 +73,7 @@ class TranslationManager:
notebook_translator=None,
translation_types: list[str] = None,
add_disclaimer: bool = True,
lang_subdir: Path | None = None,
):
"""Initialize translation manager with required components and settings.
@@ -100,6 +108,7 @@ class TranslationManager:
translation_types = ["markdown", "notebook", "images"]
self.translation_types = translation_types
self.add_disclaimer = add_disclaimer
self.lang_subdir = Path(lang_subdir) if lang_subdir else None
self.directory_manager = DirectoryManager(
root_dir,
translations_dir,
@@ -108,6 +117,17 @@ class TranslationManager:
image_dir=image_dir,
)
def _get_language_root(self, language_code: str) -> Path:
"""Return the root directory for a specific language.
Default layout is translations_dir / language_code. When lang_subdir is
set, we append it, yielding translations_dir / language_code / lang_subdir.
"""
lang_dir = self.translations_dir / language_code
if self.lang_subdir:
lang_dir = lang_dir / self.lang_subdir
return lang_dir
async def translate_image(
self, image_path: Path, language_code: str, fast_mode: bool = False
) -> str:
@@ -172,7 +192,7 @@ class TranslationManager:
document = read_input_file(file_path)
if not document:
relative_path = file_path.relative_to(self.root_dir)
output_file = self.translations_dir / language_code / relative_path
output_file = self._get_language_root(language_code) / relative_path
handle_empty_document(file_path, output_file)
return str(output_file)
@@ -214,7 +234,7 @@ class TranslationManager:
)
relative_path = file_path.relative_to(self.root_dir)
translated_path = self.translations_dir / language_code / relative_path
translated_path = self._get_language_root(language_code) / relative_path
translated_path.parent.mkdir(parents=True, exist_ok=True)
try:
@@ -224,7 +244,7 @@ class TranslationManager:
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
# Save centralized text metadata for this source file in the language directory
lang_dir = self.translations_dir / language_code
lang_dir = self._get_language_root(language_code)
save_text_metadata_for_source(
lang_dir,
file_path,
@@ -257,7 +277,7 @@ class TranslationManager:
document = read_input_file(file_path)
if not document:
relative_path = file_path.relative_to(self.root_dir)
output_file = self.translations_dir / language_code / relative_path
output_file = self._get_language_root(language_code) / relative_path
handle_empty_document(file_path, output_file)
return str(output_file)
@@ -276,7 +296,7 @@ class TranslationManager:
return ""
relative_path = file_path.relative_to(self.root_dir)
translated_path = self.translations_dir / language_code / relative_path
translated_path = self._get_language_root(language_code) / relative_path
translated_path.parent.mkdir(parents=True, exist_ok=True)
try:
@@ -286,7 +306,7 @@ class TranslationManager:
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
# Save centralized text metadata for this source notebook in the language directory
lang_dir = self.translations_dir / language_code
lang_dir = self._get_language_root(language_code)
save_text_metadata_for_source(
lang_dir,
file_path,
@@ -322,7 +342,7 @@ class TranslationManager:
if update:
for language_code in self.language_codes:
delete_translated_markdown_files_by_language_code(
language_code, self.translations_dir
language_code, self.translations_dir, self.lang_subdir
)
logger.info(
f"Deleted all translated markdown files for language: {language_code}"
@@ -342,7 +362,7 @@ class TranslationManager:
for language_code in self.language_codes:
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_path = (
self.translations_dir / language_code / relative_path
self._get_language_root(language_code) / relative_path
)
if not update and translated_md_path.exists():
@@ -404,7 +424,7 @@ class TranslationManager:
if update:
for language_code in self.language_codes:
# Find and delete translated notebook files
translation_dir = self.translations_dir / language_code
translation_dir = self._get_language_root(language_code)
if translation_dir.exists():
for ext in self.supported_notebook_extensions:
for notebook_file in translation_dir.rglob(f"*{ext}"):
@@ -425,7 +445,7 @@ class TranslationManager:
for language_code in self.language_codes:
relative_path = notebook_file_path.relative_to(self.root_dir)
translated_notebook_path = (
self.translations_dir / language_code / relative_path
self._get_language_root(language_code) / relative_path
)
if translated_notebook_path.exists() and not update:
@@ -648,6 +668,10 @@ class TranslationManager:
try:
rename_map: dict[str, str] = {}
migrated_image_count = 0
should_migrate_links = (
"markdown" in self.translation_types
or "notebook" in self.translation_types
)
if "images" in self.translation_types:
# Migrate legacy translated image filenames and update markdown/notebook links
@@ -670,10 +694,15 @@ class TranslationManager:
else:
rename_map.update(webp_rename_map)
migrated_image_count += len(webp_rename_map)
else:
logger.info(
"Skipping translated image migration because image translation is disabled"
)
if should_migrate_links:
try:
# Always run link migration to rewrite legacy flattened links in content,
# even when no files were moved (empty rename_map)
# even when no files were moved or image translation is disabled.
migrated_md = self.directory_manager.migrate_markdown_image_links(
rename_map
)
@@ -681,15 +710,15 @@ class TranslationManager:
rename_map
)
logger.info(
"Migrated %d image files and updated %d markdown and %d notebook files",
migrated_image_count,
"Updated image links in %d markdown and %d notebook files (migrated image files: %d)",
migrated_md,
migrated_nb,
migrated_image_count,
)
except Exception as e:
logger.warning(f"Image link migration skipped: {e}")
# As a safety net, canonicalize any remaining alias-based language dir segments in links
# As a safety net, canonicalize any remaining alias-based language dir segments in links.
try:
md_fix, nb_fix = canonicalize_image_links_in_translations(
self.translations_dir, self.image_dir
@@ -702,42 +731,6 @@ class TranslationManager:
)
except Exception as e:
logger.warning(f"Image link canonicalization skipped: {e}")
else:
logger.info(
"Skipping translated image migration because image translation is disabled"
)
try:
# Always run link migration to rewrite legacy flattened links in content,
# even when no files were moved (empty rename_map)
migrated_md = self.directory_manager.migrate_markdown_image_links(
rename_map
)
migrated_nb = self.directory_manager.migrate_notebook_image_links(
rename_map
)
logger.info(
"Migrated %d image files and updated %d markdown and %d notebook files",
migrated_image_count,
migrated_md,
migrated_nb,
)
except Exception as e:
logger.warning(f"Image link migration skipped: {e}")
# As a safety net, canonicalize any remaining alias-based language dir segments in links
try:
md_fix, nb_fix = canonicalize_image_links_in_translations(
self.translations_dir, self.image_dir
)
if md_fix or nb_fix:
logger.info(
"Canonicalized image links in %d markdown and %d notebooks",
md_fix,
nb_fix,
)
except Exception as e:
logger.warning(f"Image link canonicalization skipped: {e}")
# Clean up files no longer needed in target directories
logger.info("Removing orphaned files...")
@@ -809,6 +802,23 @@ class TranslationManager:
check_progress.update(1)
if outdated_files:
try:
est_tokens = estimate_tokens_for_outdated(
self,
outdated_files,
content_type="markdown",
) + estimate_tokens_for_outdated(
self,
outdated_files,
content_type="notebook",
)
logger.info(
f"Estimated tokens for selected retranslation targets: {est_tokens:,}"
)
except Exception as e:
logger.debug(
f"Failed to estimate tokens for outdated files: {e}"
)
await self.retranslate_outdated_files(outdated_files)
# Find outdated images needing retranslation
@@ -835,6 +845,14 @@ class TranslationManager:
# Execute translation for markdown, notebook and image files
if "markdown" in self.translation_types:
try:
md_pending = self._gather_pending_markdown(update=update)
md_tokens = estimate_tokens_for_sources(md_pending)
logger.info(
f"Estimated tokens for markdown translations: {md_tokens:,} (files: {len(md_pending)})"
)
except Exception as e:
logger.debug(f"Failed to estimate markdown tokens: {e}")
md_modified, md_errors = await self.translate_all_markdown_files(
update=update
)
@@ -842,6 +860,14 @@ class TranslationManager:
all_errors.extend(md_errors)
if "notebook" in self.translation_types:
try:
nb_pending = self._gather_pending_notebooks(update=update)
nb_tokens = estimate_tokens_for_sources(nb_pending)
logger.info(
f"Estimated tokens for notebook translations: {nb_tokens:,} (files: {len(nb_pending)})"
)
except Exception as e:
logger.debug(f"Failed to estimate notebook tokens: {e}")
nb_modified, nb_errors = await self.translate_all_notebook_files(
update=update
)
@@ -914,6 +940,50 @@ class TranslationManager:
return total_modified, all_errors
def estimate_tokens(self, update: bool = False) -> dict:
"""Estimate tokens for the upcoming translation run.
Backward-compatible shim that delegates token-estimation breakdown
calculation to shared estimation utilities.
"""
return estimate_translation_tokens(self, update=update)
def _gather_pending_markdown(self, update: bool) -> List[Path]:
pending: List[Path] = []
markdown_files = filter_files(self.root_dir, self.excluded_dirs)
for md_file_path in markdown_files:
md_file_path = md_file_path.resolve()
if md_file_path.suffix.lower() in SUPPORTED_MARKDOWN_EXTENSIONS:
for language_code in self.language_codes:
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_path = (
self._get_language_root(language_code) / relative_path
)
if not update and translated_md_path.exists():
continue
pending.append(md_file_path)
return pending
def _gather_pending_notebooks(self, update: bool) -> List[Path]:
pending: List[Path] = []
notebook_files: List[Path] = []
for ext in self.supported_notebook_extensions:
notebook_files.extend(filter_files(self.root_dir, self.excluded_dirs, ext))
for notebook_file_path in notebook_files:
notebook_file_path = notebook_file_path.resolve()
for language_code in self.language_codes:
relative_path = notebook_file_path.relative_to(self.root_dir)
translated_notebook_path = (
self._get_language_root(language_code) / relative_path
)
if translated_notebook_path.exists() and not update:
# Existing notebook translations are handled separately by the
# outdated-translation pass, so the pending bucket should
# only include notebooks that do not have a translation yet.
continue
pending.append(notebook_file_path)
return pending
def get_outdated_translations(self) -> List[tuple[Path, Path]]:
"""Identify translations that need updates based on file hash comparison.
@@ -926,7 +996,9 @@ class TranslationManager:
all_translation_files = []
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
translation_dir = TranslationManager._resolve_language_root(
self, lang_code
)
if not translation_dir.exists():
continue
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
@@ -941,9 +1013,8 @@ class TranslationManager:
for lang_code, trans_file in all_translation_files:
try:
relative_path = trans_file.relative_to(
self.translations_dir / lang_code
)
lang_dir = TranslationManager._resolve_language_root(self, lang_code)
relative_path = trans_file.relative_to(lang_dir)
original_file = self.root_dir / relative_path
if not original_file.exists():
@@ -958,6 +1029,21 @@ class TranslationManager:
return outdated_files
def _resolve_language_root(self, language_code: str) -> Path:
"""Return a concrete language root even for partially mocked managers in tests."""
try:
lang_dir = self._get_language_root(language_code)
if isinstance(lang_dir, Path):
return lang_dir
except Exception:
pass
lang_dir = Path(self.translations_dir) / language_code
lang_subdir = getattr(self, "lang_subdir", None)
if isinstance(lang_subdir, (str, Path)) and str(lang_subdir):
lang_dir = lang_dir / Path(lang_subdir)
return lang_dir
async def retranslate_outdated_files(
self, outdated_files: List[tuple[Path, Path]]
) -> None:
@@ -1195,7 +1281,7 @@ class TranslationManager:
# Find the path of the translated file
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_file_path = (
self.translations_dir / language_code / relative_path
self._get_language_root(language_code) / relative_path
)
if not translated_md_file_path.exists():
@@ -1354,12 +1440,29 @@ class TranslationManager:
if translation_file.suffix.lower() in self.supported_notebook_extensions:
return not is_notebook_up_to_date(original_file, translation_file)
# Determine language directory from translation path
# Determine language directory and language code from translation path
lang_dir = None
lang_code = None
try:
rel = translation_file.resolve().relative_to(self.translations_dir)
lang_code = rel.parts[0]
lang_dir = self.translations_dir / lang_code
# Find which language this file belongs to by checking roots
for lc in self.language_codes:
root = self._get_language_root(lc)
try:
# Use resolve() to handle potential symlinks or relative path complexities
rel = translation_file.resolve().relative_to(root.resolve())
lang_code = lc
lang_dir = root
break
except (ValueError, IndexError):
continue
if not lang_dir:
# Fallback to translations_dir / lang_code if not found in subdirs
rel = translation_file.resolve().relative_to(
self.translations_dir.resolve()
)
lang_code = rel.parts[0]
lang_dir = self.translations_dir / lang_code
except Exception:
# Fallback: use the parent directory (may be incorrect for deeply nested paths)
lang_dir = translation_file.parent
+67
View File
@@ -0,0 +1,67 @@
from typing import Iterable
_glossary_terms: list[str] = []
def normalize_glossary_terms(glossary_terms: Iterable[str] | None) -> list[str]:
"""Normalize glossary terms into a de-duplicated ordered list of strings."""
if not glossary_terms:
return []
normalized: list[str] = []
seen: set[str] = set()
for term in glossary_terms:
if term is None:
continue
text = str(term).strip()
if not text or text in seen:
continue
normalized.append(text)
seen.add(text)
return normalized
def set_glossary_terms(glossary_terms: Iterable[str] | None) -> None:
"""Set the process-wide glossary terms used by prompt builders."""
global _glossary_terms
_glossary_terms = normalize_glossary_terms(glossary_terms)
def get_glossary_terms() -> list[str]:
"""Return the current process-wide glossary terms."""
return list(_glossary_terms)
def _build_glossary_lines() -> list[str]:
terms = get_glossary_terms()
if not terms:
return []
lines = [
"GLOSSARY (do not translate or change these terms; keep them exactly as written):"
]
lines.extend(f"- {term}" for term in terms)
return lines
def inject_markdown_glossary(prompt: str, delimiter: str) -> str:
"""Inject glossary instructions into a markdown prompt if terms are configured."""
glossary_lines = _build_glossary_lines()
if not glossary_lines:
return prompt
glossary_block = "\n\n" + "\n".join(glossary_lines) + "\n"
index = prompt.find(delimiter)
if index == -1:
return prompt + glossary_block
prefix = prompt[:index]
suffix = prompt[index:]
return prefix + glossary_block + suffix
def build_image_glossary_block() -> str:
"""Return an image-prompt glossary block if terms are configured."""
glossary_lines = _build_glossary_lines()
if not glossary_lines:
return ""
return "\n".join(glossary_lines)
+39 -25
View File
@@ -83,24 +83,13 @@ def load_other_courses_template() -> str:
return ""
def update_readme_languages_table(
readme_path: Path, repo_url: str | None = None
) -> bool:
"""
Update README languages table between markers using bundled template.
Optionally appends a 'Prefer to Clone Locally?' advisory block INSIDE the markers.
If repo_url is provided, it will be used to personalize the snippet; otherwise
the advisory will be shown with placeholder stars as given by the user.
Returns True if updated, False otherwise.
"""
if not readme_path.exists():
return False
original = readme_path.read_text(encoding="utf-8")
def render_updated_readme_languages_table(
readme_text: str, repo_url: str | None = None
) -> str:
"""Return README content with the bundled languages table rendered."""
template = load_languages_table_template()
if not template:
return False
return readme_text
# Strip markdownlint directives from template to avoid injecting them into user README
template = re.sub(
r"^\s*<!--\s*markdownlint-disable[^>]*-->\s*\n?",
@@ -150,9 +139,37 @@ def update_readme_languages_table(
inner_content = template[start_idx + len(LANG_TABLE_START) : end_idx].strip()
new_block = f"{LANG_TABLE_START}\n{inner_content}\n{LANG_TABLE_END}"
updated = _replace_between_markers_generic(
original, new_block, LANG_TABLE_START, LANG_TABLE_END
return _replace_between_markers_generic(
readme_text, new_block, LANG_TABLE_START, LANG_TABLE_END
)
def render_updated_readme_other_courses(readme_text: str) -> str:
"""Return README content with the bundled Other courses block rendered."""
template = load_other_courses_template()
if not template:
return readme_text
return _replace_between_markers_generic(
readme_text, template, OTHER_COURSES_START, OTHER_COURSES_END
)
def update_readme_languages_table(
readme_path: Path, repo_url: str | None = None
) -> bool:
"""
Update README languages table between markers using bundled template.
Optionally appends a 'Prefer to Clone Locally?' advisory block INSIDE the markers.
If repo_url is provided, it will be used to personalize the snippet; otherwise
the advisory will be shown with placeholder stars as given by the user.
Returns True if updated, False otherwise.
"""
if not readme_path.exists():
return False
original = readme_path.read_text(encoding="utf-8")
updated = render_updated_readme_languages_table(original, repo_url=repo_url)
if updated != original:
readme_path.write_text(updated, encoding="utf-8", newline="\n")
return True
@@ -167,12 +184,7 @@ def update_readme_other_courses(readme_path: Path) -> bool:
if not readme_path.exists():
return False
original = readme_path.read_text(encoding="utf-8")
template = load_other_courses_template()
if not template:
return False
updated = _replace_between_markers_generic(
original, template, OTHER_COURSES_START, OTHER_COURSES_END
)
updated = render_updated_readme_other_courses(original)
if updated != original:
readme_path.write_text(updated, encoding="utf-8", newline="\n")
return True
@@ -828,7 +840,7 @@ def delete_translated_images_by_language_code(language_code: str, image_dir: Pat
def delete_translated_markdown_files_by_language_code(
language_code: str, translations_dir: Path
language_code: str, translations_dir: Path, lang_subdir: Path | None = None
):
"""
Delete the entire directory for the specified language code, including all its contents.
@@ -839,6 +851,8 @@ def delete_translated_markdown_files_by_language_code(
"""
# Construct the path to the directory for the specific language
language_dir = translations_dir / language_code
if lang_subdir:
language_dir = language_dir / Path(lang_subdir)
if not language_dir.exists():
logger.warning(
@@ -0,0 +1,269 @@
from pathlib import Path
from typing import Any, Literal
import logging
import tiktoken
from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS
from co_op_translator.config.llm_config.azure_openai import AzureOpenAIConfig
from co_op_translator.config.llm_config.config import LLMConfig
from co_op_translator.config.llm_config.openai import OpenAIConfig
from co_op_translator.config.llm_config.provider import LLMProvider
from co_op_translator.utils.common.file_utils import (
filter_files,
generate_translated_filename,
get_filename_and_extension,
read_input_file,
)
logger = logging.getLogger(__name__)
def _resolve_model_name() -> str | None:
"""Resolve the best-effort model name for tokenizer selection."""
try:
provider = LLMConfig.get_available_provider()
if provider == LLMProvider.AZURE_OPENAI:
return AzureOpenAIConfig.get_model_name() or None
if provider == LLMProvider.OPENAI:
return OpenAIConfig.get_chat_model_id()
except Exception:
return None
return None
def _get_encoding() -> tiktoken.Encoding | None:
model = _resolve_model_name()
try:
if model:
return tiktoken.encoding_for_model(model)
except Exception:
pass
try:
return tiktoken.get_encoding("cl100k_base")
except Exception:
return None
_ENCODING: tiktoken.Encoding | None = None
def count_tokens(text: str) -> int:
"""Return token count for the given text using best-effort encoding."""
if not text:
return 0
global _ENCODING
if _ENCODING is None:
_ENCODING = _get_encoding()
if _ENCODING is None:
return max(1, len(text) // 4)
try:
return len(_ENCODING.encode(text))
except Exception as e:
logger.debug(f"Tokenization failed, falling back to heuristic: {e}")
return max(1, len(text) // 4)
def _get_estimation_source_text(
file_path: Path,
virtual_file_contents: dict[Path, str] | None = None,
) -> str:
if virtual_file_contents:
try:
resolved_path = file_path.resolve()
except Exception:
resolved_path = file_path
if resolved_path in virtual_file_contents:
return virtual_file_contents[resolved_path].strip()
return read_input_file(file_path)
def estimate_tokens_for_sources(
files: list[Path],
virtual_file_contents: dict[Path, str] | None = None,
) -> int:
total = 0
for file_path in files:
try:
text = _get_estimation_source_text(
file_path,
virtual_file_contents=virtual_file_contents,
)
try:
total += count_tokens(text)
except Exception:
total += len(text.split())
except Exception:
continue
return total
def estimate_tokens_for_images(translation_manager: Any, update: bool) -> int:
count = 0
image_files = filter_files(
translation_manager.root_dir,
translation_manager.excluded_dirs,
)
for image_file_path in image_files:
image_file_path = Path(image_file_path).resolve()
_, ext = get_filename_and_extension(image_file_path)
if ext not in translation_manager.supported_image_extensions:
continue
for language_code in translation_manager.language_codes:
translated_filename = generate_translated_filename(
image_file_path,
language_code,
translation_manager.root_dir,
)
translated_image_path = (
Path(translation_manager.image_dir) / language_code / translated_filename
)
if not update and translated_image_path.exists():
continue
count += 1
return count * 10
def _collect_outdated_translations(
translation_manager: Any,
update: bool,
) -> list[tuple[Path, Path]]:
if not update:
return list(translation_manager.get_outdated_translations())
files: list[tuple[Path, Path]] = []
for lang_code in translation_manager.language_codes:
translation_dir = translation_manager._get_language_root(lang_code)
if not translation_dir.exists():
continue
trans_files: list[Path] = []
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
trans_files.extend(translation_dir.rglob(f"*{ext}"))
for ext in translation_manager.supported_notebook_extensions:
trans_files.extend(translation_dir.rglob(f"*{ext}"))
for trans_file in trans_files:
try:
rel = trans_file.relative_to(translation_dir)
original = translation_manager.root_dir / rel
if original.exists():
files.append((original, trans_file))
except Exception:
continue
return files
def estimate_tokens_for_outdated(
translation_manager: Any,
outdated_files: list[tuple[Path, Path]],
content_type: Literal["markdown", "notebook"],
virtual_file_contents: dict[Path, str] | None = None,
) -> int:
if content_type == "markdown":
allowed_extensions = SUPPORTED_MARKDOWN_EXTENSIONS
else:
allowed_extensions = translation_manager.supported_notebook_extensions
sources = [
original
for original, _ in outdated_files
if original.suffix.lower() in allowed_extensions
]
return estimate_tokens_for_sources(
sources,
virtual_file_contents=virtual_file_contents,
)
def estimate_tokens_for_outdated_images(
translation_manager: Any,
outdated_images: list[tuple[Path, Path, str]] | None = None,
) -> int:
try:
images = (
outdated_images
if outdated_images is not None
else translation_manager.get_outdated_images()
)
return len(images) * 10
except Exception:
return 0
def estimate_translation_tokens(
translation_manager: Any,
update: bool = False,
virtual_file_contents: dict[Path, str] | None = None,
) -> dict[str, int]:
"""Return token estimate breakdown from translation manager state."""
breakdown = {
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"markdown": 0,
"notebook": 0,
"images": 0,
}
if ("markdown" in translation_manager.translation_types) or (
"notebook" in translation_manager.translation_types
):
outdated = _collect_outdated_translations(translation_manager, update)
if outdated:
breakdown["outdated_markdown"] = estimate_tokens_for_outdated(
translation_manager,
outdated,
"markdown",
virtual_file_contents=virtual_file_contents,
)
breakdown["outdated_notebook"] = estimate_tokens_for_outdated(
translation_manager,
outdated,
"notebook",
virtual_file_contents=virtual_file_contents,
)
if "images" in translation_manager.translation_types:
breakdown["outdated_images"] = estimate_tokens_for_outdated_images(
translation_manager,
)
if "markdown" in translation_manager.translation_types:
markdown_pending = translation_manager._gather_pending_markdown(update=update)
if markdown_pending:
breakdown["markdown"] = estimate_tokens_for_sources(
markdown_pending,
virtual_file_contents=virtual_file_contents,
)
if "notebook" in translation_manager.translation_types:
notebook_pending = translation_manager._gather_pending_notebooks(update=update)
if notebook_pending:
breakdown["notebook"] = estimate_tokens_for_sources(
notebook_pending,
virtual_file_contents=virtual_file_contents,
)
if "images" in translation_manager.translation_types:
try:
breakdown["images"] = estimate_tokens_for_images(
translation_manager,
update=update,
)
except Exception:
breakdown["images"] = 0
outdated_total = (
breakdown["outdated_markdown"]
+ breakdown["outdated_notebook"]
+ breakdown["outdated_images"]
)
total = sum(breakdown.values())
return {**breakdown, "outdated": outdated_total, "total": total}
@@ -0,0 +1,106 @@
import re
from pathlib import Path
from typing import Any
from co_op_translator.config.constants import SUPPORTED_MARKDOWN_EXTENSIONS
from co_op_translator.utils.common.file_utils import read_input_file
def _get_estimation_source_text(
file_path: Path,
virtual_file_contents: dict[Path, str] | None = None,
) -> str:
if virtual_file_contents:
try:
resolved_path = file_path.resolve()
except Exception:
resolved_path = file_path
if resolved_path in virtual_file_contents:
return virtual_file_contents[resolved_path].strip()
return read_input_file(file_path)
def _estimate_words_for_sources(
files: list[Path],
virtual_file_contents: dict[Path, str] | None = None,
) -> int:
total = 0
for file_path in files:
try:
text = _get_estimation_source_text(
file_path,
virtual_file_contents=virtual_file_contents,
)
total += len(re.findall(r"\S+", text))
except Exception:
continue
return total
def _collect_outdated_sources_for_update(translation_manager: Any) -> list[Path]:
sources: list[Path] = []
for lang_code in translation_manager.language_codes:
translation_dir = translation_manager._get_language_root(lang_code)
if not translation_dir.exists():
continue
trans_files: list[Path] = []
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
trans_files.extend(translation_dir.rglob(f"*{ext}"))
for ext in translation_manager.supported_notebook_extensions:
trans_files.extend(translation_dir.rglob(f"*{ext}"))
for trans_file in trans_files:
try:
rel = trans_file.relative_to(translation_dir)
original = translation_manager.root_dir / rel
if original.exists():
sources.append(original)
except Exception:
continue
return sources
def estimate_translation_words(
translation_manager: Any,
update: bool = False,
virtual_file_contents: dict[Path, str] | None = None,
) -> dict[str, int]:
"""Estimate direct source word counts for pre-run display."""
breakdown = {"outdated": 0, "markdown": 0, "notebook": 0, "images": 0}
if ("markdown" in translation_manager.translation_types) or (
"notebook" in translation_manager.translation_types
):
if update:
outdated_sources = _collect_outdated_sources_for_update(translation_manager)
else:
outdated_sources = [
source_file
for source_file, _ in translation_manager.get_outdated_translations()
]
if outdated_sources:
breakdown["outdated"] = _estimate_words_for_sources(
outdated_sources,
virtual_file_contents=virtual_file_contents,
)
if "markdown" in translation_manager.translation_types:
md_pending = translation_manager._gather_pending_markdown(update=update)
if md_pending:
breakdown["markdown"] = _estimate_words_for_sources(
md_pending,
virtual_file_contents=virtual_file_contents,
)
if "notebook" in translation_manager.translation_types:
nb_pending = translation_manager._gather_pending_notebooks(update=update)
if nb_pending:
breakdown["notebook"] = _estimate_words_for_sources(
nb_pending,
virtual_file_contents=virtual_file_contents,
)
total = sum(breakdown.values())
return {**breakdown, "total": total}
@@ -0,0 +1,574 @@
"""Frontmatter parsing and translation utilities.
This module provides deterministic frontmatter handling by:
1. Parsing YAML frontmatter from markdown content
2. Filtering fields based on preserve/translate configuration
3. Merging translated fields back with preserved fields
4. Reconstructing markdown with updated frontmatter
This approach ensures that technical fields (slug, id, order, etc.) are NEVER
accidentally translated by the LLM, providing 100% stability for metadata.
"""
import re
import os
import logging
from pathlib import Path
from typing import Dict, List, Tuple, Any, Optional
from urllib.parse import urlparse
import yaml
from importlib import resources
from co_op_translator.utils.llm.markdown_utils import (
build_translated_image_link,
get_translated_markdown_dir,
)
logger = logging.getLogger(__name__)
class FrontmatterConfig:
"""Manages frontmatter field translation configuration."""
def __init__(self, config_path: Optional[Path] = None):
"""Initialize frontmatter configuration.
Args:
config_path: Optional path to custom configuration file.
If None, uses the default bundled configuration.
"""
self.preserve_fields: List[str] = []
self.translate_fields: List[str] = []
self._load_config(config_path)
def _load_config(self, config_path: Optional[Path] = None) -> None:
"""Load frontmatter configuration from YAML file.
Args:
config_path: Optional path to custom configuration file
"""
try:
if config_path and config_path.exists():
with open(config_path, "r", encoding="utf-8") as f:
config = yaml.safe_load(f)
else:
# Load default bundled configuration
with (
resources.files("co_op_translator.config")
.joinpath("frontmatter_config.yml")
.open("r", encoding="utf-8") as f
):
config = yaml.safe_load(f)
if config and "frontmatter" in config:
fm_config = config["frontmatter"]
self.preserve_fields = fm_config.get("preserve", [])
self.translate_fields = fm_config.get("translate", [])
logger.debug(
f"Loaded frontmatter config: {len(self.preserve_fields)} preserve fields, "
f"{len(self.translate_fields)} translate fields"
)
else:
logger.warning(
"Frontmatter configuration is empty or invalid. Using empty field lists."
)
except Exception as e:
logger.warning(
f"Failed to load frontmatter configuration: {e}. Using empty field lists."
)
def should_preserve(self, field_name: str) -> bool:
"""Check if a field should be preserved (not translated).
Args:
field_name: Name of the frontmatter field
Returns:
True if field should be preserved, False otherwise
"""
return field_name in self.preserve_fields
def should_translate(self, field_name: str) -> bool:
"""Check if a field should be translated.
Args:
field_name: Name of the frontmatter field
Returns:
True if field should be translated, False otherwise
"""
return field_name in self.translate_fields
class FrontmatterParser:
"""Parses and reconstructs YAML frontmatter in markdown documents."""
# Regex pattern for YAML frontmatter (must be at the start of the document)
FRONTMATTER_PATTERN = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL)
def __init__(self, config: Optional[FrontmatterConfig] = None):
"""Initialize frontmatter parser.
Args:
config: Optional frontmatter configuration. If None, uses default config.
"""
self.config = config or FrontmatterConfig()
def extract_frontmatter(self, content: str) -> Tuple[Optional[Dict[str, Any]], str]:
"""Extract YAML frontmatter from markdown content.
Args:
content: Full markdown document content
Returns:
Tuple of (frontmatter_dict, body_content)
- frontmatter_dict: Parsed YAML frontmatter as dict, or None if not found
- body_content: Markdown content without frontmatter
"""
match = self.FRONTMATTER_PATTERN.match(content)
if not match:
return None, content
frontmatter_yaml = match.group(1)
body = content[match.end() :]
try:
frontmatter = yaml.safe_load(frontmatter_yaml)
if not isinstance(frontmatter, dict):
logger.warning(
f"Frontmatter is not a dictionary: {type(frontmatter)}. Treating as no frontmatter."
)
return None, content
return frontmatter, body
except yaml.YAMLError as e:
logger.warning(
f"Failed to parse frontmatter YAML: {e}. Treating as no frontmatter."
)
return None, content
def split_fields(
self, frontmatter: Dict[str, Any]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Split frontmatter fields into preserve and translate groups.
Args:
frontmatter: Full frontmatter dictionary
Returns:
Tuple of (preserve_fields, translate_fields)
- preserve_fields: Fields that should not be translated
- translate_fields: Fields that should be translated
"""
preserve = {}
translate = {}
for key, value in frontmatter.items():
if self.config.should_preserve(key):
preserve[key] = value
elif self.config.should_translate(key):
translate[key] = value
else:
# Unknown field: preserve by default for safety
logger.debug(
f"Unknown frontmatter field '{key}' not in config. Preserving by default."
)
preserve[key] = value
return preserve, translate
def merge_fields(
self, preserve_fields: Dict[str, Any], translated_fields: Dict[str, Any]
) -> Dict[str, Any]:
"""Merge preserved and translated fields back into a single frontmatter dict.
Args:
preserve_fields: Fields that were preserved (not translated)
translated_fields: Fields that were translated
Returns:
Merged frontmatter dictionary
"""
# Start with preserved fields, then add translated fields
# This ensures preserved fields take precedence in case of conflicts
merged = preserve_fields.copy()
merged.update(translated_fields)
return merged
def reconstruct_content(
self, frontmatter: Optional[Dict[str, Any]], body: str
) -> str:
"""Reconstruct markdown content with frontmatter.
Args:
frontmatter: Frontmatter dictionary (or None if no frontmatter)
body: Markdown body content
Returns:
Full markdown content with frontmatter
"""
if not frontmatter:
return body
# Serialize frontmatter to YAML
try:
frontmatter_yaml = yaml.dump(
frontmatter,
allow_unicode=True,
default_flow_style=False,
sort_keys=False,
)
return f"---\n{frontmatter_yaml}---\n{body}"
except Exception as e:
logger.error(
f"Failed to serialize frontmatter to YAML: {e}. Returning body only."
)
return body
def extract_translatable_fields_as_markdown(
self, translate_fields: Dict[str, Any]
) -> str:
"""Convert translatable fields to a markdown format for LLM translation.
This creates a simple markdown representation of the fields that the LLM
can translate naturally.
Args:
translate_fields: Dictionary of fields to translate
Returns:
Markdown-formatted string of translatable fields
"""
if not translate_fields:
return ""
lines = []
for key, value in translate_fields.items():
if isinstance(value, str):
# Simple string field
lines.append(f"**{key}**: {value}")
elif isinstance(value, list):
# List field (e.g., multiple descriptions)
lines.append(f"**{key}**:")
for item in value:
if isinstance(item, str):
lines.append(f"- {item}")
elif isinstance(value, dict):
# Nested dict (rare, but handle gracefully)
lines.append(f"**{key}**: {yaml.dump(value, allow_unicode=True)}")
else:
# Other types: convert to string
lines.append(f"**{key}**: {str(value)}")
return "\n".join(lines)
def parse_translated_fields_from_markdown(
self, translated_markdown: str, original_fields: Dict[str, Any]
) -> Dict[str, Any]:
"""Parse translated fields from LLM-generated markdown.
This attempts to extract the translated values from the markdown format
created by extract_translatable_fields_as_markdown.
Args:
translated_markdown: LLM-translated markdown content
original_fields: Original translatable fields (for structure reference)
Returns:
Dictionary of translated fields
"""
translated = {}
# Pattern to match field lines: **field_name**: value
field_pattern = re.compile(r"^\*\*(.+?)\*\*:\s*(.*)$", re.MULTILINE)
for match in field_pattern.finditer(translated_markdown):
field_name = match.group(1).strip()
field_value = match.group(2).strip()
if field_name in original_fields:
# Preserve the original type
original_value = original_fields[field_name]
if isinstance(original_value, str):
translated[field_name] = field_value
elif isinstance(original_value, list):
# For lists, we need to extract list items
# This is a simplified approach; may need enhancement
translated[field_name] = [field_value]
else:
# For other types, use string representation
translated[field_name] = field_value
return translated
class QuickLinksFrontmatterParser(FrontmatterParser):
"""Frontmatter parser that promotes quickLinks nested strings for translation."""
QUICKLINKS_FIELD_NAME = "quickLinks"
QUICKLINKS_TRANSLATABLE_FIELDS = ("title", "description")
QUICKLINKS_KEY_PATTERN = re.compile(r"^quickLinks\[(\d+)\]\.(\w+)$")
def split_fields(
self, frontmatter: Dict[str, Any]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
preserve, translate = super().split_fields(frontmatter)
quicklinks = frontmatter.get(self.QUICKLINKS_FIELD_NAME)
if isinstance(quicklinks, list):
preserve[self.QUICKLINKS_FIELD_NAME] = quicklinks
translate.update(self._collect_quicklinks_translation_fields(quicklinks))
return preserve, translate
def merge_fields(
self, preserve_fields: Dict[str, Any], translated_fields: Dict[str, Any]
) -> Dict[str, Any]:
preserve_copy = dict(preserve_fields)
translated_copy = dict(translated_fields)
self._apply_nested_field_translations(preserve_copy, translated_copy)
return super().merge_fields(preserve_copy, translated_copy)
def _collect_quicklinks_translation_fields(
self, quicklinks: List[Any]
) -> Dict[str, str]:
pseudo_fields: Dict[str, str] = {}
for index, item in enumerate(quicklinks):
if not isinstance(item, dict):
continue
for field in self.QUICKLINKS_TRANSLATABLE_FIELDS:
value = item.get(field)
if isinstance(value, str):
pseudo_key = f"{self.QUICKLINKS_FIELD_NAME}[{index}].{field}"
pseudo_fields[pseudo_key] = value
return pseudo_fields
def _apply_nested_field_translations(
self,
preserve_fields: Dict[str, Any],
translated_fields: Dict[str, Any],
) -> None:
quicklinks = preserve_fields.get(self.QUICKLINKS_FIELD_NAME)
if not isinstance(quicklinks, list):
return
keys_to_remove: List[str] = []
for key, value in list(translated_fields.items()):
match = self.QUICKLINKS_KEY_PATTERN.match(key)
if not match:
continue
index = int(match.group(1))
field_name = match.group(2)
if field_name not in self.QUICKLINKS_TRANSLATABLE_FIELDS:
keys_to_remove.append(key)
continue
if 0 <= index < len(quicklinks):
item = quicklinks[index]
if isinstance(item, dict):
item[field_name] = value
keys_to_remove.append(key)
for key in keys_to_remove:
translated_fields.pop(key, None)
# Singleton instance for global use
_default_parser: Optional[FrontmatterParser] = None
def get_frontmatter_parser(
config_path: Optional[Path] = None,
) -> FrontmatterParser:
"""Get or create the default frontmatter parser instance.
Args:
config_path: Optional path to custom configuration file
Returns:
FrontmatterParser instance
"""
global _default_parser
if _default_parser is None or config_path is not None:
_default_parser = QuickLinksFrontmatterParser(FrontmatterConfig(config_path))
return _default_parser
def ensure_quicklinks_frontmatter_parser() -> QuickLinksFrontmatterParser:
"""Ensure the global parser supports translating quickLinks nested fields."""
global _default_parser
parser = get_frontmatter_parser()
if isinstance(parser, QuickLinksFrontmatterParser):
return parser
config = getattr(parser, "config", None)
_default_parser = QuickLinksFrontmatterParser(config=config)
return _default_parser
# List of frontmatter fields that typically contain file paths or URLs
PATH_FIELDS = [
"image",
"cover",
"thumbnail",
"featured_image",
"og_image",
"twitter_image",
"icon",
"canonical_url",
"url",
"permalink",
]
def adjust_frontmatter_links(
frontmatter: Dict[str, Any],
md_file_path: Path,
language_code: str,
root_dir: Path,
translations_dir: Path,
translated_images_dir: Path,
translation_types: List[str],
lang_subdir: Path | None = None,
) -> Dict[str, Any]:
"""Adjust file paths in frontmatter fields to point to correct locations.
This follows the same logic as markdown-only mode: links point to original
files, with relative paths adjusted based on the translated file's location.
Args:
frontmatter: Frontmatter dictionary with potential file paths
md_file_path: Path to the original markdown file
language_code: Target language code
root_dir: Root directory of the project
translations_dir: Directory containing translations
translated_images_dir: Directory containing translated images
translation_types: List of file types being translated
Returns:
Frontmatter dictionary with adjusted paths
"""
if not frontmatter:
return frontmatter
adjusted = frontmatter.copy()
use_translated_images = "images" in translation_types
# Calculate translated markdown directory
try:
translated_md_dir = get_translated_markdown_dir(
md_file_path,
language_code,
translations_dir,
root_dir,
lang_subdir=lang_subdir,
)
except ValueError:
logger.warning(
f"Cannot calculate relative path for '{md_file_path}' from root '{root_dir}'. "
f"Skipping frontmatter link adjustment."
)
return adjusted
for field in PATH_FIELDS:
if field not in adjusted:
continue
value = adjusted[field]
if not isinstance(value, str):
continue
# Skip web URLs and email addresses
parsed_url = urlparse(value)
if (
parsed_url.scheme in ("mailto", "http", "https")
or "@" in value
or value.endswith((".com", ".org", ".net"))
):
logger.debug(f"Skipping web URL in frontmatter field '{field}': {value}")
continue
path = parsed_url.path
if not path:
continue
# Determine if this is an image field
is_image_field = any(
img_keyword in field.lower()
for img_keyword in ["image", "cover", "thumbnail", "icon"]
)
try:
if path.startswith("/"):
# Root-relative path
if is_image_field and use_translated_images:
# For images with translation enabled, point to translated images
# This requires resolving the actual image path
actual_image_path = root_dir / path.lstrip("/")
if actual_image_path.exists():
adjusted[field] = build_translated_image_link(
path,
md_file_path,
language_code,
translated_md_dir,
translated_images_dir,
root_dir,
)
logger.debug(
f"Adjusted root-relative image path in '{field}': {value} -> {adjusted[field]}"
)
else:
# Keep original if file doesn't exist
logger.debug(
f"Root-relative path not found, keeping original in '{field}': {value}"
)
else:
# For non-images or when not using translated images, keep root-relative path
logger.debug(f"Keeping root-relative path in '{field}': {value}")
else:
# Regular relative path
original_linked_file_path = (md_file_path.parent / path).resolve()
if is_image_field and use_translated_images:
# Point to translated image
if original_linked_file_path.exists():
adjusted[field] = build_translated_image_link(
path,
md_file_path,
language_code,
translated_md_dir,
translated_images_dir,
root_dir,
)
logger.debug(
f"Adjusted relative image path in '{field}': {value} -> {adjusted[field]}"
)
else:
# Fallback to relative path to original
adjusted[field] = os.path.relpath(
original_linked_file_path, translated_md_dir
).replace(os.path.sep, "/")
logger.debug(
f"Image not found, using relative path to original in '{field}': {adjusted[field]}"
)
else:
# Point to original file (markdown-only mode behavior)
adjusted[field] = os.path.relpath(
original_linked_file_path, translated_md_dir
).replace(os.path.sep, "/")
logger.debug(
f"Adjusted relative path to original in '{field}': {value} -> {adjusted[field]}"
)
except Exception as e:
logger.warning(
f"Failed to adjust path in frontmatter field '{field}': {value}. Error: {e}. "
f"Keeping original value."
)
# Keep original value on error
return adjusted
+146 -127
View File
@@ -25,6 +25,7 @@ from co_op_translator.utils.common.file_utils import (
map_original_to_translated,
)
from co_op_translator.utils.common.lang_utils import normalize_language_code
from co_op_translator.glossary import inject_markdown_glossary
logger = logging.getLogger(__name__)
@@ -225,41 +226,76 @@ def generate_prompt_template(
language_code: str, language_name: str, document_chunk: str, is_rtl: bool
) -> str:
"""
Generate a translation prompt for a document chunk, considering language direction.
Args:
language_code (str): The target language code for translation.
language_name (str): The target language name for translation.
document_chunk (str): The chunk of the document to be translated.
is_rtl (bool): Whether the target language is right-to-left.
Returns:
str: The generated translation prompt.
Generate a safe and stable translation prompt that enforces strict
markdown structure preservation and prevents HTML/markdown rewriting.
"""
# --- 1) ONE-LINE CHUNKS (simple text or inline cases) ---
if len(document_chunk.split("\n")) == 1:
prompt = f"Translate the following text to {language_name} ({language_code}). NEVER ADD ANY EXTRA CONTENT OR TAGS OUTSIDE THE TRANSLATION. DO NOT ADD '''markdown OR ANY OTHER TAGS. TRANSLATE ONLY WHAT IS GIVEN TO YOU. MAINTAIN MARKDOWN FORMAT."
prompt = (
f"Translate the following text to {language_name} ({language_code}). "
"STRICT RULE: Do NOT add, remove, or modify any markdown characters. "
"Do NOT introduce HTML tags. Translate ONLY text content. "
"Return ONLY the translation."
)
# --- 2) MULTI-LINE CHUNKS (markdown documents) ---
else:
prompt = f"""
Translate the following markdown file to {language_name} ({language_code}).
IMPORTANT RULES:
1. DO NOT add '''markdown or any other tags around the translation
2. Make sure the translation does not sound too literal
3. Translate comments as well
4. Preserve inline and block HTML (e.g., <a>, <img>, <details>, <summary>, <div>) exactly; do not convert to Markdown; translate only visible text (e.g., link text, alt/title, summary labels); do not change tag names, attributes, or URLs/paths.
5. Do not translate:
- [!NOTE], [!WARNING], [!TIP], [!IMPORTANT], [!CAUTION]
- Variable names, function names, class names
- Placeholders like @@INLINE_CODE_x@@ or @@CODE_BLOCK_x@@
- URLs or paths
6. Keep all original markdown formatting intact
7. Return ONLY the translated content without any additional tags or markup
"""
Translate the following markdown file to {language_name} ({language_code}).
STRICT RULES (NO EXCEPTIONS):
0. STRUCTURE PRESERVATION
- You MUST preserve the exact markdown syntax from the input.
- DO NOT canonicalize, optimize, reformat, or rewrite markdown.
- DO NOT convert markdown links ([text](url)) into HTML <a> tags.
- DO NOT convert existing HTML into markdown.
- DO NOT introduce any new HTML tags.
1. OUTPUT FORMAT
- Return ONLY the translated content.
- Do NOT add ```markdown or ANY wrappers.
- Do NOT add explanations, metadata, or comments.
2. TRANSLATION SCOPE
- Translate written, human-readable text ONLY.
- DO NOT translate:
* URLs or file paths
* Markdown syntax
* Variable names, function names, class names
* Placeholders like @@INLINE_CODE_x@@ and @@CODE_BLOCK_x@@
* Tags such as [!NOTE], [!TIP], [!WARNING], [!IMPORTANT], [!CAUTION]
3. HTML HANDLING RULES
- Preserve all HTML EXACTLY as provided. This includes inline and block elements
such as <a>, <img>, <details>, <summary>, <div>, and any other HTML tags.
- DO NOT alter tag names, attributes, URLs, paths, classes, IDs, or structure.
- Translate ONLY the visible human-readable text content (e.g., link text, alt/title text,
<summary> labels, descriptive text inside tags).
- DO NOT convert HTML to Markdown or Markdown to HTML.
- DO NOT add, remove, reorder, or rewrite ANY HTML. Maintain exact byte-for-byte
fidelity for all tags and attributes.
4. FRONTMATTER RULE (YAML delimited by '---')
- If a YAML frontmatter block exists at the top of the document:
* KEEP the entire block EXACTLY.
* DO NOT modify field names or file paths.
* Translate ONLY human-readable values (e.g., title, description).
* DO NOT “fix” or normalize layout/import paths.
5. SAFETY
- Do NOT reorder lines.
- Do NOT remove blank lines.
- Do NOT merge or split paragraphs.
- Preserve whitespace, indentation, and list structure exactly.
"""
# Direction rule (minimal + unambiguous)
if is_rtl:
prompt += "Please write the output from right to left, respecting that this is a right-to-left language.\n"
prompt += "\nWrite the output in right-to-left direction.\n"
else:
prompt += "Please write the output from left to right.\n"
prompt += "\nWrite the output in left-to-right direction.\n"
language_template = _read_language_prompt_template(language_code)
if language_template:
@@ -269,7 +305,7 @@ def generate_prompt_template(
prompt += SPLIT_DELIMITER
prompt += document_chunk
return prompt
return inject_markdown_glossary(prompt, SPLIT_DELIMITER)
def get_tokenizer(encoding_name: str):
@@ -606,6 +642,60 @@ def update_links(
return markdown_string
def get_translated_markdown_dir(
md_file_path: Path,
language_code: str,
translations_dir: Path,
root_dir: Path,
lang_subdir: Path | None = None,
) -> Path:
"""Return the directory containing the translated markdown file."""
language_root = translations_dir / language_code
if lang_subdir:
language_root = language_root / Path(lang_subdir)
try:
_ = md_file_path.relative_to(language_root)
return md_file_path.parent.resolve()
except Exception:
return (language_root / md_file_path.relative_to(root_dir).parent).resolve()
def build_translated_image_link(
path: str,
md_file_path: Path,
language_code: str,
translated_md_dir: Path,
translated_images_dir: Path,
root_dir: Path,
) -> str:
"""Build a translated image link relative to the translated markdown file."""
base_names = {
translated_images_dir.name,
"translated_images",
"translated_images_fast",
}
parts = path.split("/")
rel_path = os.path.relpath(translated_images_dir.resolve(), translated_md_dir)
if len(parts) >= 3 and parts[-3] in base_names and parts[-2] == language_code:
return os.path.join(rel_path, language_code, parts[-1]).replace(
os.path.sep, "/"
)
if path.startswith("/"):
actual_image_path = get_actual_image_path(path, md_file_path, root_dir)
else:
actual_image_path = get_actual_image_path(path, md_file_path)
new_filename = generate_translated_filename(
actual_image_path, language_code, root_dir
)
return os.path.join(rel_path, language_code, new_filename).replace(
os.path.sep, "/"
)
def _slugify_heading_text(text: str) -> str:
"""Create a GitHub-style anchor slug from heading text."""
text = re.sub(r"`([^`]*)`", r"\1", text)
@@ -1005,15 +1095,12 @@ def update_image_links(
# Target translated markdown directory structure: translations/<lang>/<relative_path_to_parent>
# The translated file will be saved at: translations_dir / language_code / (md_file_path relative to root_dir)
# Its directory is:
try:
_ = md_file_path.relative_to(translations_dir / language_code)
translated_md_dir = md_file_path.parent.resolve()
except Exception:
translated_md_dir = (
translations_dir
/ language_code
/ md_file_path.relative_to(root_dir).parent
).resolve()
translated_md_dir = get_translated_markdown_dir(
md_file_path,
language_code,
translations_dir,
root_dir,
)
if not use_translated_images:
# Link to original image when using original images
@@ -1038,51 +1125,14 @@ def update_image_links(
logger.info(f"Using original image link: {updated_link}")
else:
try:
base_names = {
translated_images_dir.name,
"translated_images",
"translated_images_fast",
}
parts = path.split("/")
if (
len(parts) >= 3
and parts[-3] in base_names
and parts[-2] == language_code
):
rel_path = os.path.relpath(
translated_images_dir.resolve(), translated_md_dir
)
updated_link = os.path.join(
rel_path,
language_code,
parts[-1],
).replace(os.path.sep, "/")
else:
# Pass root_dir to get_actual_image_path to properly handle root-relative paths
if path.startswith("/"):
# For root-relative paths, we need to use the root_dir
logger.info(
f"Root-relative path detected in non-markdown-only mode: {path}"
)
# Use the modified get_actual_image_path that accepts root_dir
actual_image_path = get_actual_image_path(
path, md_file_path, root_dir
)
else:
# No change for regular paths
actual_image_path = get_actual_image_path(
path, md_file_path
)
rel_path = os.path.relpath(
translated_images_dir.resolve(), translated_md_dir
)
new_filename = generate_translated_filename(
actual_image_path, language_code, root_dir
)
updated_link = os.path.join(
rel_path, language_code, new_filename
).replace(os.path.sep, "/")
updated_link = build_translated_image_link(
path,
md_file_path,
language_code,
translated_md_dir,
translated_images_dir,
root_dir,
)
logger.info(f"Using translated image link: {updated_link}")
except Exception as e:
logger.error(f"Error processing image path {path}: {e}")
@@ -1122,15 +1172,12 @@ def update_image_links(
return match.group(0)
try:
try:
_ = md_file_path.relative_to(translations_dir / language_code)
translated_md_dir = md_file_path.parent.resolve()
except Exception:
translated_md_dir = (
translations_dir
/ language_code
/ md_file_path.relative_to(root_dir).parent
).resolve()
translated_md_dir = get_translated_markdown_dir(
md_file_path,
language_code,
translations_dir,
root_dir,
)
if not use_translated_images:
# Link to original image when using original images
@@ -1142,42 +1189,14 @@ def update_image_links(
original_linked_file_path, translated_md_dir
).replace(os.path.sep, "/")
else:
base_names = {
translated_images_dir.name,
"translated_images",
"translated_images_fast",
}
parts = path.split("/")
if (
len(parts) >= 3
and parts[-3] in base_names
and parts[-2] == language_code
):
rel_path = os.path.relpath(
translated_images_dir.resolve(), translated_md_dir
)
updated_src = os.path.join(
rel_path,
language_code,
parts[-1],
).replace(os.path.sep, "/")
else:
if path.startswith("/"):
actual_image_path = get_actual_image_path(
path, md_file_path, root_dir
)
else:
actual_image_path = get_actual_image_path(path, md_file_path)
rel_path = os.path.relpath(
translated_images_dir.resolve(), translated_md_dir
)
new_filename = generate_translated_filename(
actual_image_path, language_code, root_dir
)
updated_src = os.path.join(
rel_path, language_code, new_filename
).replace(os.path.sep, "/")
updated_src = build_translated_image_link(
path,
md_file_path,
language_code,
translated_md_dir,
translated_images_dir,
root_dir,
)
except Exception as e:
logger.error(f"Error processing HTML <img> path {src}: {e}")
updated_src = src
+19 -3
View File
@@ -7,6 +7,7 @@ import re
import logging
from typing import List
from pydantic import BaseModel
from co_op_translator.glossary import build_image_glossary_block
class TranslationResponse(BaseModel):
@@ -46,16 +47,31 @@ def gen_image_translation_prompt(text_data, language_code, language_name):
str: Generated translation prompt for structured output.
"""
line_count = len(text_data)
numbered_lines = "\n".join(f"{i}. {line}" for i, line in enumerate(text_data, 1))
numbered_lines = "\n".join(f"[{i}] {line}" for i, line in enumerate(text_data, 1))
glossary_block = build_image_glossary_block()
glossary_section = ""
if glossary_block:
glossary_section = (
"\n\nGLOSSARY INSTRUCTIONS:\n"
"- The glossary section is instruction-only metadata, not input text.\n"
"- Do not add glossary terms unless they already appear in an input line.\n"
"- If a glossary term appears in an input line, preserve it exactly as written.\n"
f"{glossary_block}"
)
prompt = f"""Translate to {language_name} ({language_code}). Return EXACTLY {line_count} items.
RULES:
- Translate ONLY the numbered INPUT LINES section
- Do NOT translate or echo glossary instructions themselves
- Output translated text only, without line numbers
- Keep symbols/numbers unchanged: +, -, →, 123
- Empty input → empty string ""
- Empty input → empty string ""{glossary_section}
{numbered_lines}"""
INPUT LINES:
{numbered_lines}
END INPUT LINES"""
return prompt
@@ -393,3 +393,42 @@ async def test_translate_markdown_full_integration(real_markdown_translator, tmp
assert (
"[Default Translation]" in result
), "Expected the default translation text in the output."
def test_insert_metadata_comment_after_frontmatter(tmp_path):
translator = ConcreteMarkdownTranslator(root_dir=tmp_path)
content_with_frontmatter = """---
layout: ../layouts/DocsLayout.astro
title: Co-op Translator - Quick Start Guide
---
# Heading
Body
"""
metadata = {
"original_hash": "hash",
"translation_date": "2025-10-15T03:44:55+00:00",
"source_file": "README.md",
"language_code": "sw",
}
metadata_comment = translator.format_metadata_comment(metadata)
result = translator._insert_metadata_comment(
content_with_frontmatter, metadata_comment
)
lines = result.splitlines()
assert lines[0] == "---"
assert lines[1].startswith("layout:")
assert lines[2].startswith("title:")
assert lines[3] == "---"
# There should be a blank line after frontmatter, then the metadata comment
assert lines[4] == ""
assert lines[5] == "<!--"
# CO_OP_TRANSLATOR_METADATA label should appear inside the comment block
assert any("CO_OP_TRANSLATOR_METADATA:" in line for line in lines)
@@ -0,0 +1,151 @@
import json
from pathlib import Path
from co_op_translator.core.project.translation_manager import TranslationManager
from co_op_translator.config.constants import (
EXCLUDED_DIRS,
SUPPORTED_IMAGE_EXTENSIONS,
SUPPORTED_NOTEBOOK_EXTENSIONS,
)
from co_op_translator.utils.common.token_estimation import count_tokens
from co_op_translator.utils.common.file_utils import generate_translated_filename
def _make_manager(
root: Path, languages: list[str], types: list[str]
) -> TranslationManager:
translations_dir = root / "translations"
image_dir = root / "translated_images"
translations_dir.mkdir(parents=True, exist_ok=True)
image_dir.mkdir(parents=True, exist_ok=True)
# We don't use translators in estimation; pass minimal placeholders
markdown_translator = None # type: ignore
image_translator = None
notebook_translator = None
return TranslationManager(
root,
translations_dir,
image_dir,
languages,
EXCLUDED_DIRS,
SUPPORTED_IMAGE_EXTENSIONS,
SUPPORTED_NOTEBOOK_EXTENSIONS,
markdown_translator, # type: ignore
image_translator,
notebook_translator,
types,
)
def test_estimate_tokens_markdown_only(tmp_path: Path):
root = tmp_path
# Create a simple markdown file
md = root / "README.md"
content = "Hello world! This is a small test."
md.write_text(content, encoding="utf-8")
mgr = _make_manager(root, ["ko", "ja"], ["markdown"]) # two languages
est = mgr.estimate_tokens(update=False)
expected = count_tokens(content) * 2
assert est["markdown"] == expected
assert est["notebook"] == 0
assert est["outdated_markdown"] == 0
assert est["outdated_notebook"] == 0
assert est["outdated_images"] == 0
assert est["outdated"] == 0
assert est["total"] == expected
def test_estimate_tokens_notebook_only(tmp_path: Path):
root = tmp_path
# Create a simple notebook (.ipynb) file with some text in cells
nb = root / "example.ipynb"
nb_content = {
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": ["Notebook example text for token estimation."],
}
],
"metadata": {},
"nbformat": 4,
"nbformat_minor": 5,
}
nb.write_text(json.dumps(nb_content), encoding="utf-8")
mgr = _make_manager(root, ["fr", "de", "es"], ["notebook"]) # three languages
est = mgr.estimate_tokens(update=False)
# Use raw file text because that's what read_input_file() reads
text = nb.read_text(encoding="utf-8")
expected = count_tokens(text) * 3
assert est["markdown"] == 0
assert est["notebook"] == expected
assert est["outdated_markdown"] == 0
assert est["outdated_notebook"] == 0
assert est["outdated_images"] == 0
assert est["outdated"] == 0
assert est["total"] == expected
def test_estimate_tokens_splits_outdated_by_content_type(tmp_path: Path):
root = tmp_path
md = root / "guide.md"
md_content = "Outdated markdown source content"
md.write_text(md_content, encoding="utf-8")
nb = root / "lesson.ipynb"
nb_content = {
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": ["Outdated notebook source content"],
}
],
"metadata": {},
"nbformat": 4,
"nbformat_minor": 5,
}
nb.write_text(json.dumps(nb_content), encoding="utf-8")
img = root / "diagram.png"
img.write_bytes(b"fake-image")
mgr = _make_manager(root, ["ko"], ["markdown", "notebook", "images"])
translated_md = root / "translations" / "ko" / "guide.md"
translated_md.parent.mkdir(parents=True, exist_ok=True)
translated_md.write_text("translated", encoding="utf-8")
translated_nb = root / "translations" / "ko" / "lesson.ipynb"
translated_nb.parent.mkdir(parents=True, exist_ok=True)
translated_nb.write_text(json.dumps(nb_content), encoding="utf-8")
translated_img_name = generate_translated_filename(img, "ko", root)
translated_img = root / "translated_images" / "ko" / translated_img_name
translated_img.parent.mkdir(parents=True, exist_ok=True)
translated_img.write_bytes(b"translated-image")
est = mgr.estimate_tokens(update=False)
expected_outdated_markdown = count_tokens(md_content)
expected_outdated_notebook = count_tokens(nb.read_text(encoding="utf-8"))
assert est["markdown"] == 0
assert est["notebook"] == 0
assert est["images"] == 0
assert est["outdated_markdown"] == expected_outdated_markdown
assert est["outdated_notebook"] == expected_outdated_notebook
assert est["outdated_images"] == 10
assert est["outdated"] == (
expected_outdated_markdown + expected_outdated_notebook + 10
)
assert est["total"] == est["outdated"]
+312
View File
@@ -0,0 +1,312 @@
from unittest.mock import MagicMock
import pytest
from co_op_translator.api import translation as api
@pytest.mark.asyncio
async def test_run_translation_calls_project_translator(tmp_path):
root_dir = tmp_path
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
api.run_translation(
language_codes="ko ja",
root_dir=str(root_dir),
update=False,
images=False,
markdown=True,
notebook=False,
debug=False,
save_logs=False,
yes=True,
)
assert project_translator_class.call_count == 2
project_translator_class.assert_any_call(
"ko ja",
str(root_dir),
translation_types=["markdown"],
add_disclaimer=False,
translations_dir=None,
image_dir=None,
lang_subdir=None,
)
project_translator_instance.translate_project.assert_called_once_with(
update=False,
)
@pytest.mark.asyncio
async def test_run_translation_with_disclaimer_flag(tmp_path):
root_dir = tmp_path
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
api.run_translation(
language_codes="ko",
root_dir=str(root_dir),
markdown=True,
add_disclaimer=True,
)
assert project_translator_class.call_count == 2
project_translator_class.assert_any_call(
"ko",
str(root_dir),
translation_types=["markdown"],
add_disclaimer=True,
translations_dir=None,
image_dir=None,
lang_subdir=None,
)
@pytest.mark.asyncio
async def test_run_translation_with_multiple_root_dirs(tmp_path):
root1 = tmp_path / "content1"
root2 = tmp_path / "content2"
root1.mkdir()
root2.mkdir()
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
api.run_translation(
language_codes="ko",
markdown=True,
root_dirs=[str(root1), str(root2)],
)
assert project_translator_class.call_count == 4
called_roots = {call.args[1] for call in project_translator_class.call_args_list}
assert called_roots == {str(root1), str(root2)}
@pytest.mark.asyncio
async def test_run_translation_with_groups(tmp_path):
root1 = tmp_path / "content1"
root2 = tmp_path / "content2"
out1 = tmp_path / "out1"
out2 = tmp_path / "out2"
root1.mkdir()
root2.mkdir()
out1.mkdir()
out2.mkdir()
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
groups = [
(str(root1), str(out1)),
(str(root2), str(out2)),
]
api.run_translation(
language_codes="ko",
markdown=True,
groups=groups,
)
assert project_translator_class.call_count == 4
called = {
(call.args[1], call.kwargs["translations_dir"])
for call in project_translator_class.call_args_list
}
assert called == {(str(root1), str(out1)), (str(root2), str(out2))}
@pytest.mark.asyncio
async def test_run_translation_dry_run_groups_shows_single_aggregated_estimate(
tmp_path,
):
root1 = tmp_path / "content1"
root2 = tmp_path / "content2"
out1 = tmp_path / "out1"
out2 = tmp_path / "out2"
root1.mkdir()
root2.mkdir()
out1.mkdir()
out2.mkdir()
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
translators = [MagicMock() for _ in range(4)]
api.ProjectTranslator = MagicMock(side_effect=translators)
api.estimate_translation_tokens = MagicMock(
side_effect=[
{
"markdown": 65,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 65,
},
{
"markdown": 65,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 65,
},
]
)
api.estimate_translation_words = MagicMock(
side_effect=[
{
"markdown": 40,
"notebook": 0,
"images": 0,
"outdated": 0,
"total": 40,
},
{
"markdown": 40,
"notebook": 0,
"images": 0,
"outdated": 0,
"total": 40,
},
]
)
echo_mock = MagicMock()
api.click.echo = echo_mock
groups = [
(str(root1), str(out1)),
(str(root2), str(out2)),
]
api.run_translation(
language_codes="ko",
markdown=True,
groups=groups,
dry_run=True,
)
estimate_lines = [
call.args[0]
for call in echo_mock.call_args_list
if call.args
and "Estimated translation volume before translation" in call.args[0]
]
assert len(estimate_lines) == 1
grouped_progress_lines = [
call.args[0]
for call in echo_mock.call_args_list
if call.args and "Translating all groups" in call.args[0]
]
assert grouped_progress_lines == []
assert (
estimate_lines[0]
== "📊 Estimated translation volume before translation: 130 tokens (80 words) "
"(breakdown: translation: markdown: 130 | retranslation: outdated markdowns: 0)"
)
@pytest.mark.asyncio
async def test_run_translation_dry_run_uses_virtual_readme_without_writing(tmp_path):
readme_path = tmp_path / "README.md"
readme_path.write_text(
"\n".join(
[
"# Sample",
"",
"<!-- CO-OP TRANSLATOR LANGUAGES TABLE START -->",
"old languages",
"<!-- CO-OP TRANSLATOR LANGUAGES TABLE END -->",
"",
"<!-- CO-OP TRANSLATOR OTHER COURSES START -->",
"old courses",
"<!-- CO-OP TRANSLATOR OTHER COURSES END -->",
"",
]
),
encoding="utf-8",
)
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
api.update_readme_languages_table = MagicMock(return_value=True)
api.update_readme_other_courses = MagicMock(return_value=True)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
token_estimate_mock = MagicMock(
return_value={
"markdown": 10,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 10,
}
)
word_estimate_mock = MagicMock(
return_value={
"markdown": 5,
"notebook": 0,
"images": 0,
"outdated": 0,
"total": 5,
}
)
api.estimate_translation_tokens = token_estimate_mock
api.estimate_translation_words = word_estimate_mock
original_readme = readme_path.read_text(encoding="utf-8")
api.run_translation(
language_codes="ko",
root_dir=str(tmp_path),
markdown=True,
dry_run=True,
repo_url="https://github.com/example/repo",
)
virtual_inputs = token_estimate_mock.call_args.kwargs["virtual_file_contents"]
assert readme_path.resolve() in virtual_inputs
assert virtual_inputs[readme_path.resolve()] != original_readme
assert "old languages" not in virtual_inputs[readme_path.resolve()]
assert "old courses" not in virtual_inputs[readme_path.resolve()]
assert readme_path.read_text(encoding="utf-8") == original_readme
api.update_readme_languages_table.assert_not_called()
api.update_readme_other_courses.assert_not_called()
@@ -0,0 +1,584 @@
"""Tests for frontmatter parsing and translation utilities."""
import pytest
from pathlib import Path
from co_op_translator.utils.llm.frontmatter_utils import (
FrontmatterConfig,
FrontmatterParser,
QuickLinksFrontmatterParser,
ensure_quicklinks_frontmatter_parser,
get_frontmatter_parser,
adjust_frontmatter_links,
)
class TestFrontmatterConfig:
"""Test frontmatter configuration loading and field classification."""
def test_load_default_config(self):
"""Test loading default bundled configuration."""
config = FrontmatterConfig()
# Should have preserve fields
assert len(config.preserve_fields) > 0
assert "slug" in config.preserve_fields
assert "id" in config.preserve_fields
assert "order" in config.preserve_fields
assert "section" in config.preserve_fields
# Should have translate fields
assert len(config.translate_fields) > 0
assert "title" in config.translate_fields
assert "description" in config.translate_fields
def test_should_preserve(self):
"""Test field preservation check."""
config = FrontmatterConfig()
assert config.should_preserve("slug") is True
assert config.should_preserve("id") is True
assert config.should_preserve("order") is True
assert config.should_preserve("title") is False
def test_should_translate(self):
"""Test field translation check."""
config = FrontmatterConfig()
assert config.should_translate("title") is True
assert config.should_translate("description") is True
assert config.should_translate("slug") is False
assert config.should_translate("id") is False
class TestFrontmatterParser:
"""Test frontmatter parsing and reconstruction."""
def test_extract_frontmatter_with_valid_yaml(self):
"""Test extracting valid YAML frontmatter."""
content = """---
title: Getting Started
slug: getting-started
section: introduction
order: 1
---
# Welcome
This is the body content.
"""
parser = FrontmatterParser()
frontmatter, body = parser.extract_frontmatter(content)
assert frontmatter is not None
assert frontmatter["title"] == "Getting Started"
assert frontmatter["slug"] == "getting-started"
assert frontmatter["section"] == "introduction"
assert frontmatter["order"] == 1
assert body.strip().startswith("# Welcome")
def test_extract_frontmatter_without_frontmatter(self):
"""Test extracting from content without frontmatter."""
content = """# Welcome
This is just body content.
"""
parser = FrontmatterParser()
frontmatter, body = parser.extract_frontmatter(content)
assert frontmatter is None
assert body == content
def test_extract_frontmatter_with_invalid_yaml(self):
"""Test extracting invalid YAML frontmatter."""
content = """---
title: Getting Started
invalid yaml here: [unclosed bracket
---
# Welcome
"""
parser = FrontmatterParser()
frontmatter, body = parser.extract_frontmatter(content)
# Should treat as no frontmatter on parse error
assert frontmatter is None
assert body == content
def test_split_fields(self):
"""Test splitting frontmatter fields into preserve and translate."""
frontmatter = {
"title": "Getting Started",
"description": "Welcome to our docs",
"slug": "getting-started",
"section": "introduction",
"order": 1,
}
parser = FrontmatterParser()
preserve, translate = parser.split_fields(frontmatter)
# Preserve fields
assert "slug" in preserve
assert "section" in preserve
assert "order" in preserve
assert preserve["slug"] == "getting-started"
assert preserve["order"] == 1
# Translate fields
assert "title" in translate
assert "description" in translate
assert translate["title"] == "Getting Started"
assert translate["description"] == "Welcome to our docs"
def test_split_fields_unknown_field_preserved(self):
"""Test that unknown fields are preserved by default."""
frontmatter = {
"title": "Getting Started",
"unknown_field": "some value",
}
parser = FrontmatterParser()
preserve, translate = parser.split_fields(frontmatter)
# Unknown field should be preserved for safety
assert "unknown_field" in preserve
assert preserve["unknown_field"] == "some value"
def test_merge_fields(self):
"""Test merging preserved and translated fields."""
preserve = {
"slug": "getting-started",
"section": "introduction",
"order": 1,
}
translate = {
"title": "시작하기",
"description": "문서에 오신 것을 환영합니다",
}
parser = FrontmatterParser()
merged = parser.merge_fields(preserve, translate)
assert merged["slug"] == "getting-started"
assert merged["section"] == "introduction"
assert merged["order"] == 1
assert merged["title"] == "시작하기"
assert merged["description"] == "문서에 오신 것을 환영합니다"
def test_reconstruct_content_with_frontmatter(self):
"""Test reconstructing content with frontmatter."""
frontmatter = {
"title": "시작하기",
"description": "문서에 오신 것을 환영합니다",
"slug": "getting-started",
"section": "introduction",
"order": 1,
}
body = "# 환영합니다\n\n번역된 내용입니다."
parser = FrontmatterParser()
result = parser.reconstruct_content(frontmatter, body)
assert result.startswith("---\n")
assert "title: 시작하기" in result
assert "slug: getting-started" in result
assert "order: 1" in result
assert "# 환영합니다" in result
def test_reconstruct_content_without_frontmatter(self):
"""Test reconstructing content without frontmatter."""
body = "# Welcome\n\nBody content."
parser = FrontmatterParser()
result = parser.reconstruct_content(None, body)
assert result == body
def test_extract_translatable_fields_as_markdown(self):
"""Test converting translatable fields to markdown format."""
translate_fields = {
"title": "Getting Started",
"description": "Welcome to our documentation",
}
parser = FrontmatterParser()
markdown = parser.extract_translatable_fields_as_markdown(translate_fields)
assert "**title**: Getting Started" in markdown
assert "**description**: Welcome to our documentation" in markdown
def test_parse_translated_fields_from_markdown(self):
"""Test parsing translated fields from markdown."""
original_fields = {
"title": "Getting Started",
"description": "Welcome to our documentation",
}
translated_markdown = """**title**: 시작하기
**description**: 문서에 오신 것을 환영합니다"""
parser = FrontmatterParser()
translated = parser.parse_translated_fields_from_markdown(
translated_markdown, original_fields
)
assert "title" in translated
assert "description" in translated
assert translated["title"] == "시작하기"
assert translated["description"] == "문서에 오신 것을 환영합니다"
def test_end_to_end_frontmatter_workflow(self):
"""Test complete frontmatter translation workflow."""
original_content = """---
title: Getting Started
description: Welcome to our docs
slug: getting-started
section: introduction
order: 1
---
# Welcome
This is the body content.
"""
parser = FrontmatterParser()
# Step 1: Extract frontmatter
frontmatter, body = parser.extract_frontmatter(original_content)
assert frontmatter is not None
# Step 2: Split fields
preserve, translate = parser.split_fields(frontmatter)
assert "slug" in preserve
assert "title" in translate
# Step 3: Simulate translation (in real scenario, this goes to LLM)
translated_fields = {
"title": "시작하기",
"description": "문서에 오신 것을 환영합니다",
}
# Step 4: Merge fields
merged = parser.merge_fields(preserve, translated_fields)
# Step 5: Reconstruct content
translated_body = "# 환영합니다\n\n번역된 내용입니다."
result = parser.reconstruct_content(merged, translated_body)
# Verify result
assert "---" in result
assert "title: 시작하기" in result
assert "slug: getting-started" in result # Preserved
assert "section: introduction" in result # Preserved
assert "order: 1" in result # Preserved
assert "# 환영합니다" in result
class TestGetFrontmatterParser:
"""Test singleton parser getter."""
def test_get_default_parser(self):
"""Test getting default parser instance."""
parser = get_frontmatter_parser()
assert parser is not None
assert isinstance(parser, FrontmatterParser)
assert isinstance(parser, QuickLinksFrontmatterParser)
def test_singleton_behavior(self):
"""Test that parser is reused (singleton pattern)."""
parser1 = get_frontmatter_parser()
parser2 = get_frontmatter_parser()
assert parser1 is parser2
def test_ensure_quicklinks_parser_overrides_default(self):
parser = ensure_quicklinks_frontmatter_parser()
assert isinstance(parser, QuickLinksFrontmatterParser)
again = get_frontmatter_parser()
assert again is parser
class TestQuickLinksFrontmatterParser:
def test_split_fields_promotes_nested_strings(self):
parser = QuickLinksFrontmatterParser()
frontmatter = {
"title": "Getting Started",
"quickLinks": [
{
"slug": "getting-started",
"title": "Getting Started",
"description": "Learn the basics",
},
{
"slug": "advanced",
"title": "Advanced",
"description": "Go deeper",
},
],
}
preserve, translate = parser.split_fields(frontmatter)
assert "quickLinks" in preserve
assert preserve["quickLinks"][0]["slug"] == "getting-started"
assert translate["quickLinks[0].title"] == "Getting Started"
assert translate["quickLinks[0].description"] == "Learn the basics"
assert translate["quickLinks[1].title"] == "Advanced"
assert translate["quickLinks[1].description"] == "Go deeper"
def test_merge_fields_reinserts_quicklinks_translations(self):
parser = QuickLinksFrontmatterParser()
preserve = {
"quickLinks": [
{
"slug": "getting-started",
"title": "Getting Started",
"description": "Learn the basics",
},
{
"slug": "advanced",
"title": "Advanced",
"description": "Go deeper",
},
]
}
translated = {
"quickLinks[0].title": "시작하기",
"quickLinks[0].description": "기본을 배우세요",
"quickLinks[1].title": "고급",
"quickLinks[1].description": "더 깊이 탐구하세요",
}
merged = parser.merge_fields(preserve, translated)
assert merged["quickLinks"][0]["title"] == "시작하기"
assert merged["quickLinks"][0]["description"] == "기본을 배우세요"
assert merged["quickLinks"][0]["slug"] == "getting-started"
assert merged["quickLinks"][1]["title"] == "고급"
assert merged["quickLinks"][1]["description"] == "더 깊이 탐구하세요"
class TestAdjustFrontmatterLinks:
"""Test frontmatter link adjustment functionality."""
def test_adjust_relative_image_path_markdown_only(self, tmp_path):
"""Test adjusting relative image path in markdown-only mode."""
# Setup directory structure
root_dir = tmp_path / "project"
root_dir.mkdir()
docs_dir = root_dir / "docs"
docs_dir.mkdir()
images_dir = root_dir / "images"
images_dir.mkdir()
# Create a dummy image file
image_file = images_dir / "hero.png"
image_file.write_text("dummy")
# Create markdown file
md_file = docs_dir / "guide.md"
md_file.write_text("# Guide")
# Frontmatter with relative image path
frontmatter = {
"title": "Getting Started",
"image": "../images/hero.png",
"slug": "getting-started",
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
# Adjust links (markdown-only mode: images not in translation_types)
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown"], # No images
)
# Should point to original image with adjusted relative path
assert "image" in adjusted
# From translations/ko/docs/guide.md to images/hero.png
assert adjusted["image"] == "../../../images/hero.png"
# Other fields should be unchanged
assert adjusted["title"] == "Getting Started"
assert adjusted["slug"] == "getting-started"
def test_adjust_root_relative_path(self, tmp_path):
"""Test adjusting root-relative path."""
root_dir = tmp_path / "project"
root_dir.mkdir()
docs_dir = root_dir / "docs"
docs_dir.mkdir()
md_file = docs_dir / "guide.md"
md_file.write_text("# Guide")
frontmatter = {
"title": "Guide",
"canonical_url": "/docs/guide",
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown"],
)
# Root-relative paths should be kept as-is
assert adjusted["canonical_url"] == "/docs/guide"
def test_skip_web_urls(self, tmp_path):
"""Test that web URLs are not modified."""
root_dir = tmp_path / "project"
root_dir.mkdir()
md_file = root_dir / "README.md"
md_file.write_text("# README")
frontmatter = {
"title": "Project",
"og_image": "https://example.com/image.png",
"canonical_url": "https://example.com/docs",
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown"],
)
# Web URLs should remain unchanged
assert adjusted["og_image"] == "https://example.com/image.png"
assert adjusted["canonical_url"] == "https://example.com/docs"
def test_adjust_with_translated_images(self, tmp_path):
"""Test adjusting image paths when using translated images."""
root_dir = tmp_path / "project"
root_dir.mkdir()
docs_dir = root_dir / "docs"
docs_dir.mkdir()
images_dir = root_dir / "images"
images_dir.mkdir()
image_file = images_dir / "hero.png"
image_file.write_text("dummy")
md_file = docs_dir / "guide.md"
md_file.write_text("# Guide")
frontmatter = {
"title": "Getting Started",
"image": "../images/hero.png",
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
# Adjust links with images in translation_types
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown", "images"], # Images enabled
)
# Should point to translated image directory
assert "image" in adjusted
# Path should point to translated_images with language-specific filename
assert "translated_images" in adjusted["image"]
assert "/ko/" in adjusted["image"] or adjusted["image"].startswith("../ko/")
def test_adjust_with_translated_images_and_lang_subdir(self, tmp_path):
"""Test translated image paths remain correct when lang_subdir is used."""
root_dir = tmp_path / "project"
root_dir.mkdir()
docs_dir = root_dir / "docs"
docs_dir.mkdir()
images_dir = root_dir / "images"
images_dir.mkdir()
image_file = images_dir / "hero.png"
image_file.write_text("dummy")
md_file = docs_dir / "guide.md"
md_file.write_text("# Guide")
frontmatter = {
"title": "Getting Started",
"image": "../images/hero.png",
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown", "images"],
lang_subdir=Path("docs"),
)
assert adjusted["image"].startswith("../../../../translated_images/ko/")
assert adjusted["image"].endswith(".webp")
def test_no_modification_for_non_path_fields(self, tmp_path):
"""Test that non-path fields are not modified."""
root_dir = tmp_path / "project"
root_dir.mkdir()
md_file = root_dir / "README.md"
md_file.write_text("# README")
frontmatter = {
"title": "Project",
"description": "A great project",
"order": 1,
"tags": ["python", "translation"],
}
translations_dir = root_dir / "translations"
translated_images_dir = root_dir / "translated_images"
adjusted = adjust_frontmatter_links(
frontmatter,
md_file,
"ko",
root_dir,
translations_dir,
translated_images_dir,
translation_types=["markdown"],
)
# All fields should remain unchanged (no path fields)
assert adjusted == frontmatter
@@ -1,6 +1,7 @@
import pytest
import os
from pathlib import Path
from co_op_translator.glossary import set_glossary_terms
from co_op_translator.utils.llm.markdown_utils import (
update_links,
@@ -177,6 +178,16 @@ def test_generate_prompt_template_without_language_template_for_non_configured_l
assert "STRUCTURE IS MORE IMPORTANT THAN STYLE." not in prompt
def test_generate_prompt_template_includes_glossary_when_configured():
try:
set_glossary_terms(["Co-op Translator"])
prompt = generate_prompt_template("ko", "Korean", "Test content", False)
assert "GLOSSARY" in prompt
assert "Co-op Translator" in prompt
finally:
set_glossary_terms([])
def test_count_links_in_markdown():
"""Test counting links in markdown content."""
content = """
@@ -4,6 +4,7 @@ from co_op_translator.utils.llm.text_utils import (
strip_line_number_prefix,
TranslationResponse,
)
from co_op_translator.glossary import set_glossary_terms
def test_remove_code_backticks():
@@ -35,9 +36,9 @@ def test_gen_image_translation_prompt():
assert "EXACTLY 3 items" in prompt
assert "without line numbers" in prompt
# Check numbered lines are included
assert "1. Line 1" in prompt
assert "2. Line 2" in prompt
assert "3. Line 3" in prompt
assert "[1] Line 1" in prompt
assert "[2] Line 2" in prompt
assert "[3] Line 3" in prompt
def test_strip_line_number_prefix():
@@ -78,6 +79,23 @@ def test_gen_image_translation_prompt_special_chars():
assert all(line in prompt for line in text_data)
def test_gen_image_translation_prompt_includes_glossary_and_keeps_rules():
try:
set_glossary_terms(["Co-op Translator"])
prompt = gen_image_translation_prompt(["Co-op Translator Cloud"], "ko", "Korean")
assert "GLOSSARY" in prompt
assert "Co-op Translator" in prompt
assert "EXACTLY 1 items" in prompt
assert "without line numbers" in prompt
assert "instruction-only metadata" in prompt
assert "Translate ONLY the numbered INPUT LINES section" in prompt
assert "INPUT LINES:" in prompt
assert "END INPUT LINES" in prompt
assert "[1] Co-op Translator Cloud" in prompt
finally:
set_glossary_terms([])
def test_translation_response():
"""Test TranslationResponse Pydantic model."""
translations = ["Translated line 1", "Translated line 2", "Translated line 3"]