Core: Standardize language codes to canonical BCP 47 across CLI, core, templates, and metadata (#350)

This commit is contained in:
Minseok Song
2026-01-28 21:08:13 +09:00
committed by GitHub
parent 5d44c6a3b4
commit 7587708098
22 changed files with 992 additions and 120 deletions
+5 -4
View File
@@ -1,9 +1,9 @@
# 🌐 Multi-Language Support (Template)
Maintainers: The block below is an "all languages" example managed by Co‑op Translator.
Maintainers: The block below is an "all languages" example managed by Co-op Translator.
- If you want Co‑op Translator to keep this list fully up‑to‑date automatically when you run `translate` (any language selection), keep the two comment markers as-is.
- If you only want to show a subset of languages, delete the two comment markers and remove any languages you don't want to list. After removing the markers, Co‑op Translator will no longer auto‑replace this section.
- If you want Co-op Translator to keep this list fully up-to-date automatically when you run `translate` (any language selection), keep the two comment markers as-is.
- If you only want to show a subset of languages, delete the two comment markers and remove any languages you don't want to list. After removing the markers, Co-op Translator will no longer auto-replace this section.
- The section now includes a "Prefer to Clone Locally?" advisory to help users clone without the large translations payload. You can personalize the advisory with your repository URL by running, for example:
- `translate -l "ko" --repo-url "https://github.com/org/repo.git"`
@@ -15,7 +15,7 @@ Maintainers: The block below is an "all languages" example managed by Co‑op Tr
#### Supported by [Co-op Translator](https://github.com/Azure/Co-op-Translator)
<!-- CO-OP TRANSLATOR LANGUAGES TABLE START -->
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh/README.md) | [Chinese (Traditional, Hong Kong)](./translations/hk/README.md) | [Chinese (Traditional, Macau)](./translations/mo/README.md) | [Chinese (Traditional, Taiwan)](./translations/tw/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/br/README.md) | [Portuguese (Portugal)](./translations/pt/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh-CN/README.md) | [Chinese (Traditional, Hong Kong)](./translations/zh-HK/README.md) | [Chinese (Traditional, Macau)](./translations/zh-MO/README.md) | [Chinese (Traditional, Taiwan)](./translations/zh-TW/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/pt-BR/README.md) | [Portuguese (Portugal)](./translations/pt-PT/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
> **Prefer to Clone Locally?**
>
> This repository includes 50+ language translations which significantly increases the download size. To clone without translations, use sparse checkout:
@@ -29,3 +29,4 @@ Maintainers: The block below is an "all languages" example managed by Co‑op Tr
<!-- CO-OP TRANSLATOR LANGUAGES TABLE END -->
```
+6 -6
View File
@@ -12,10 +12,10 @@ The table below lists the languages currently supported by **Co-op Translator**.
| ar | Arabic | NotoSansArabic-Medium.ttf | Yes | No |
| fa | Persian (Farsi) | NotoSansArabic-Medium.ttf | Yes | No |
| ur | Urdu | NotoSansArabic-Medium.ttf | Yes | No |
| zh | Chinese (Simplified) | NotoSansCJK-Medium.ttc | No | No |
| mo | Chinese (Traditional, Macau) | NotoSansCJK-Medium.ttc | No | No |
| hk | Chinese (Traditional, Hong Kong) | NotoSansCJK-Medium.ttc| No | No |
| tw | Chinese (Traditional, Taiwan) | NotoSansCJK-Medium.ttc | No | No |
| zh-CN | Chinese (Simplified) | NotoSansCJK-Medium.ttc | No | No |
| zh-MO | Chinese (Traditional, Macau) | NotoSansCJK-Medium.ttc | No | No |
| zh-HK | Chinese (Traditional, Hong Kong) | NotoSansCJK-Medium.ttc| No | No |
| zh-TW | Chinese (Traditional, Taiwan) | NotoSansCJK-Medium.ttc | No | No |
| ja | Japanese | NotoSansCJK-Medium.ttc | No | No |
| ko | Korean | NotoSansCJK-Medium.ttc | No | No |
| hi | Hindi | NotoSansDevanagari-Medium.ttf | No | No |
@@ -23,8 +23,8 @@ The table below lists the languages currently supported by **Co-op Translator**.
| mr | Marathi | NotoSansDevanagari-Medium.ttf | No | No |
| ne | Nepali | NotoSansDevanagari-Medium.ttf | No | No |
| pa | Punjabi (Gurmukhi) | NotoSansGurmukhi-Medium.ttf | No | No |
| pt | Portuguese (Portugal)| NotoSans-Medium.ttf | No | No |
| br | Portuguese (Brazil) | NotoSans-Medium.ttf | No | No |
| pt-PT | Portuguese (Portugal)| NotoSans-Medium.ttf | No | No |
| pt-BR | Portuguese (Brazil) | NotoSans-Medium.ttf | No | No |
| it | Italian | NotoSans-Medium.ttf | No | No |
| lt | Lithuanian | NotoSans-Medium.ttf | No | No |
| pl | Polish | NotoSans-Medium.ttf | No | No |
+12 -9
View File
@@ -11,6 +11,7 @@ import os
from co_op_translator.core.project.project_evaluator import ProjectEvaluator
from co_op_translator.config.base_config import Config
from co_op_translator.utils.common.logging_utils import setup_logging
from co_op_translator.utils.common.lang_utils import normalize_language_code
logger = logging.getLogger(__name__)
@@ -93,7 +94,9 @@ def evaluate_command(
if save_logs and log_file_path is not None:
click.echo(f"📄 Logs will be saved to: {log_file_path}")
click.echo(f"Evaluating {language_code} translations in {root_path}...")
# Normalize to canonical BCP 47 (accept alias input like tw/cn/br)
canonical_code = normalize_language_code(language_code)
click.echo(f"Evaluating {canonical_code} translations in {root_path}...")
# Create evaluator
# Determine evaluation mode (fast, deep or default mode)
@@ -121,7 +124,7 @@ def evaluate_command(
evaluator = ProjectEvaluator(
root_dir=root_path,
translations_dir=root_path / "translations",
language_codes=[language_code],
language_codes=[canonical_code],
excluded_dirs=["node_modules", ".git", "__pycache__", "venv"],
use_llm=use_llm,
use_rule=use_rule,
@@ -129,7 +132,7 @@ def evaluate_command(
# Run evaluation
total_files, issue_files, avg_confidence = asyncio.run(
evaluator.evaluate_project(language_code)
evaluator.evaluate_project(canonical_code)
)
# Display results with color highlighting
@@ -154,7 +157,7 @@ def evaluate_command(
# Get low confidence translations
low_confidence = asyncio.run(
evaluator.get_low_confidence_translations(language_code, min_confidence)
evaluator.get_low_confidence_translations(canonical_code, min_confidence)
)
if low_confidence:
@@ -175,10 +178,10 @@ def evaluate_command(
# Check if the path already contains translations/language_code
# to avoid duplication
if rel_path.startswith(f"translations/{language_code}/"):
if rel_path.startswith(f"translations/{canonical_code}/"):
display_path = f"./{rel_path}"
else:
display_path = f"./translations/{language_code}/{rel_path}"
display_path = f"./translations/{canonical_code}/{rel_path}"
except ValueError:
display_path = str(file_path).replace("\\", "/")
@@ -190,7 +193,7 @@ def evaluate_command(
)
trans_path = Path(file_path)
lang_dir = root_path / "translations" / language_code
lang_dir = root_path / "translations" / canonical_code
try:
rel = trans_path.resolve().relative_to(lang_dir)
orig_path = root_path / rel
@@ -239,7 +242,7 @@ def evaluate_command(
f"\n{click.style('Note:', fg='yellow')} Files with issues were found during evaluation, but none fall below the confidence threshold of {min_confidence}."
)
click.echo(
f"Consider running with a higher threshold: {click.style(f'evaluate -l {language_code} --min-confidence 0.9', bold=True)}"
f"Consider running with a higher threshold: {click.style(f'evaluate -l {canonical_code} --min-confidence 0.9', bold=True)}"
)
else:
@@ -247,7 +250,7 @@ def evaluate_command(
f"\n{click.style('✓ All translations look good!', fg='green', bold=True)}"
)
logger.info(f"Evaluation completed for language: {language_code}")
logger.info(f"Evaluation completed for language: {canonical_code}")
except Exception as e:
if debug:
+33 -32
View File
@@ -13,20 +13,22 @@ import os
import re
from urllib.parse import urlparse
from tqdm import tqdm
import importlib.resources
import yaml
from co_op_translator.config.base_config import Config
from co_op_translator.utils.llm.markdown_utils import (
migrate_notebook_links,
update_notebook_links,
)
from co_op_translator.utils.common.file_utils import map_original_to_translated
from co_op_translator.utils.common.file_utils import (
map_original_to_translated,
canonicalize_image_links_in_translations,
)
from co_op_translator.utils.common.logging_utils import setup_logging
from co_op_translator.config.constants import (
SUPPORTED_MARKDOWN_EXTENSIONS,
SUPPORTED_NOTEBOOK_EXTENSIONS,
)
from co_op_translator.utils.common.lang_utils import normalize_language_codes
logger = logging.getLogger(__name__)
@@ -44,6 +46,11 @@ logger = logging.getLogger(__name__)
default=".",
help="Root directory of the project (default is current directory).",
)
@click.option(
"--image-dir",
default="translated_images",
help="Base directory for translated images (relative to --root-dir).",
)
@click.option(
"--dry-run",
is_flag=True,
@@ -74,6 +81,7 @@ logger = logging.getLogger(__name__)
def migrate_links_command(
language_codes,
root_dir,
image_dir,
dry_run,
fallback_to_original,
debug,
@@ -106,6 +114,19 @@ def migrate_links_command(
click.echo(f"No translations directory found at: {translations_dir}")
return
# Canonicalize legacy alias-based language segments in links across translated content
try:
md_fix, nb_fix = canonicalize_image_links_in_translations(
translations_dir=translations_dir,
image_dir=(root_path / image_dir),
)
if md_fix or nb_fix:
click.echo(
f"✅ Canonicalized alias language segments in links: markdown={md_fix}, notebooks={nb_fix}"
)
except Exception as e:
logger.debug(f"Canonicalization step skipped: {e}")
# Warning and confirmation when processing all languages
if isinstance(language_codes, str) and language_codes.lower() == "all":
click.echo(
@@ -129,34 +150,12 @@ def migrate_links_command(
# Parse language codes list (support "all")
if isinstance(language_codes, str) and language_codes.lower() == "all":
try:
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file)
if not font_mappings:
raise click.ClickException("Empty font mappings file")
lang_list = [
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
if not lang_list:
raise click.ClickException(
"No valid language codes found in font mappings"
)
logging.debug(
f"Expanded 'all' to language codes from font mapping: {lang_list}"
)
except (FileNotFoundError, yaml.YAMLError) as e:
raise click.ClickException(
f"Failed to load language codes for 'all': {str(e)}"
)
lang_list = Config.get_language_codes()
else:
lang_list = [
code.strip() for code in language_codes.split() if code.strip()
]
lang_list = normalize_language_codes(lang_list)
if not lang_list:
raise click.ClickException("No valid language codes provided.")
@@ -181,7 +180,9 @@ def migrate_links_command(
md_files: list[Path] = []
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
md_files.extend(lang_dir.rglob(f"*{ext}"))
for md_translated in tqdm(md_files, desc=f"{lang} md", unit="file"):
for md_translated in tqdm(
md_files, desc=f"{lang_dir.name} md", unit="file"
):
total_scanned += 1
try:
content = md_translated.read_text(encoding="utf-8")
@@ -260,7 +261,7 @@ def migrate_links_command(
# Determine translated counterpart if exists
candidate_translated = map_original_to_translated(
original_abs=linked_abs,
language_code=lang,
language_code=lang_dir.name,
root_dir=root_path,
)
@@ -278,7 +279,7 @@ def migrate_links_command(
# Compute translated_md_dir for later comparisons
translated_md_dir = (
translations_dir
/ lang
/ lang_dir.name
/ original_md_path.relative_to(root_path).parent
)
@@ -329,7 +330,7 @@ def migrate_links_command(
updated = update_notebook_links(
markdown_string=content,
md_file_path=original_md_path,
language_code=lang,
language_code=lang_dir.name,
translations_dir=translations_dir,
root_dir=root_path,
use_translated_notebook=True,
@@ -338,7 +339,7 @@ def migrate_links_command(
updated = migrate_notebook_links(
markdown_string=content,
md_file_path=original_md_path,
language_code=lang,
language_code=lang_dir.name,
root_dir=root_path,
)
+93 -27
View File
@@ -5,9 +5,6 @@ Translate command implementation for Co-op Translator CLI.
import asyncio
import logging
import click
import importlib.resources
import yaml
import os
from pathlib import Path
from co_op_translator.core.project.project_translator import ProjectTranslator
@@ -19,6 +16,13 @@ from co_op_translator.utils.common.file_utils import (
update_readme_languages_table,
update_readme_other_courses,
)
from co_op_translator.utils.common.lang_utils import (
normalize_language_codes,
)
from co_op_translator.utils.common.metadata_utils import (
normalize_language_codes_in_lang_metadata,
)
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
logger = logging.getLogger(__name__)
@@ -91,6 +95,19 @@ logger = logging.getLogger(__name__)
default=None,
help="Repository URL to show in the 'Prefer to Clone Locally?' advisory inside the languages table.",
)
@click.option(
"--migrate-language-folders",
is_flag=True,
help=(
"Detect and optionally rename alias-based language folders (e.g., tw, cn, br) "
"to canonical BCP 47 (zh-TW, zh-CN, pt-BR)."
),
)
@click.option(
"--dry-run",
is_flag=True,
help="Preview migration plan without making changes (use with --migrate-language-folders).",
)
def translate_command(
language_codes,
root_dir,
@@ -106,6 +123,8 @@ def translate_command(
min_confidence,
add_disclaimer,
repo_url,
migrate_language_folders,
dry_run,
):
"""
CLI for translating project files.
@@ -193,6 +212,8 @@ def translate_command(
if save_logs and log_file_path is not None:
click.echo(f"📄 Logs will be saved to: {log_file_path}")
# (Preview moved after language normalization)
# Now run the LLM health check; raises on failure
LLMConfig.validate_connectivity()
logger.info("LLM health check passed.")
@@ -205,7 +226,7 @@ def translate_command(
logger.info("Vision health check passed.")
click.echo("✅ Vision health check passed.")
# Show warning if 'all' is selected
# Normalize language codes and handle 'all'
all_languages_selected = language_codes == "all"
if all_languages_selected:
click.echo(
@@ -228,31 +249,76 @@ def translate_command(
click.echo("Proceeding with translation for all languages...")
else:
click.echo("Auto-confirming translation for all languages...")
# Use canonical list from config and normalize
lang_list = Config.get_language_codes()
if not lang_list:
raise click.ClickException(
"No valid language codes found in font mappings"
)
language_codes = " ".join(normalize_language_codes(lang_list))
else:
# Normalize explicit input codes to canonical form
lang_list = normalize_language_codes(
[code.strip() for code in language_codes.split()]
)
if not lang_list:
raise click.ClickException("No valid language codes provided")
language_codes = " ".join(lang_list)
try:
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file)
if not font_mappings:
raise click.ClickException("Empty font mappings file")
language_codes = " ".join(
[
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
)
if not language_codes:
raise click.ClickException(
"No valid language codes found in font mappings"
# Detect and migrate alias-based language folders for the SELECTED languages
# This runs before any translation work to avoid redundant re-translation.
try:
migrator = LanguageFolderMigrator(root_path)
alias_entries = migrator.detect_alias_folders()
if alias_entries:
# Filter only entries relevant to selected canonical languages
relevant = [e for e in alias_entries if e.canonical in lang_list]
if relevant:
click.echo("\nMigration plan (selected languages):")
click.echo(LanguageFolderMigrator.format_plan(relevant))
if dry_run:
click.echo("Dry-run: no changes will be made.")
else:
do_migrate = migrate_language_folders or yes
if not do_migrate:
# Ask for confirmation when not explicitly requested and not auto-confirmed
confirm = click.prompt(
"Migrate alias folders now? Type 'yes' to continue",
type=str,
default="no",
)
logging.debug(
f"Loaded language codes from font mapping: {language_codes}"
)
except (FileNotFoundError, yaml.YAMLError) as e:
raise click.ClickException(f"Failed to load font mappings: {str(e)}")
do_migrate = confirm.strip().lower() == "yes"
if do_migrate:
renamed, msgs = migrator.execute(relevant, dry_run=False)
click.echo(f"Auto-migrate: renamed {renamed} folder(s).")
for m in msgs:
click.echo(f"- {m}")
else:
click.echo("Proceeding without migration.")
else:
if migrate_language_folders and dry_run:
click.echo("No non-standard language folders detected.")
except Exception as e:
logger.warning(f"Language folder migration step skipped: {e}")
# Ensure per-language metadata files store canonical language_code values
try:
for lang in lang_list:
# translations/<lang>/.co-op-translator.json
normalize_language_codes_in_lang_metadata(
root_path / "translations" / lang, lang
)
# translated_images/<lang>/.co-op-translator.json
normalize_language_codes_in_lang_metadata(
root_path / "translated_images" / lang, lang
)
# translated_images_fast/<lang>/.co-op-translator.json (best-effort)
normalize_language_codes_in_lang_metadata(
root_path / "translated_images_fast" / lang, lang
)
except Exception as e:
logger.debug(f"Metadata normalization skipped: {e}")
# Show deprecation warning when fast image mode is enabled
if fast and "images" in translation_types:
+28 -7
View File
@@ -6,6 +6,7 @@ import importlib.resources
import yaml
from co_op_translator.config.llm_config.config import LLMConfig
from co_op_translator.config.vision_config.config import VisionConfig
from co_op_translator.utils.common.lang_utils import normalize_language_code
logger = logging.getLogger(__name__)
@@ -46,7 +47,8 @@ class Config:
@staticmethod
def get_language_codes() -> list[str]:
"""
Return the full list of supported language codes from the packaged font mappings.
Return the full list of supported language codes from the packaged font mappings,
normalized to canonical BCP 47 (e.g., zh-TW, zh-CN, pt-BR, pt-PT).
Falls back to an empty list on error.
"""
@@ -56,12 +58,31 @@ class Config:
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file) or {}
# Only include entries with a dict mapping (same rule as CLI)
return [
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
# Only include entries with a dict mapping and normalize
seen: set[str] = set()
ordered: list[str] = []
for key, meta in font_mappings.items():
if not isinstance(meta, dict):
continue
key_str = str(key)
canon = normalize_language_code(key_str)
# Optional developer warning when a non-canonical key is present
try:
if os.getenv("COOP_STRICT_CANON_KEYS") == "1" and (
not canon or canon != key_str
):
logger.warning(
"Non-canonical or invalid language code '%s' in font_language_mappings.yml (normalized to '%s')",
key_str,
canon,
)
except Exception:
# Do not fail on logging issues
pass
if canon and canon not in seen:
seen.add(canon)
ordered.append(canon)
return ordered
except Exception as e:
logger.warning(f"Failed to load language codes from font mappings: {e}")
return []
+32 -7
View File
@@ -1,5 +1,8 @@
import importlib.resources
import yaml
from co_op_translator.utils.common.lang_utils import (
normalize_language_code,
)
class FontConfig:
@@ -13,6 +16,23 @@ class FontConfig:
with open(mappings_path, "r", encoding="utf-8") as file:
self.font_mappings = yaml.safe_load(file)
def _resolve_mapping_key(self, language_code: str) -> str:
"""
Resolve provided language code to a canonical BCP 47 key present in font_language_mappings.yml.
The YAML now uses canonical keys only. Alias inputs are normalized before lookup.
"""
if not language_code:
raise ValueError("Empty language code is not supported.")
# Normalize input (accept alias like "tw", "cn")
canonical = normalize_language_code(language_code)
if canonical in self.font_mappings:
return canonical
raise ValueError(
f"Font for language code '{language_code}' is not supported or not found."
)
def get_font_path(self, language_code):
"""
Retrieve the font path for a given language code.
@@ -26,7 +46,8 @@ class FontConfig:
Raises:
ValueError: If the language code or font is not found in the mappings.
"""
font_name = self.font_mappings.get(language_code, {}).get("font")
key = self._resolve_mapping_key(language_code)
font_name = self.font_mappings.get(key, {}).get("font")
if not font_name:
raise ValueError(
@@ -49,10 +70,12 @@ class FontConfig:
Raises:
ValueError: If the language code is not found in the mappings.
"""
if language_code not in self.font_mappings:
try:
key = self._resolve_mapping_key(language_code)
except ValueError:
# Preserve historical error message for tests/backward-compat
raise ValueError(f"Language code '{language_code}' is not supported.")
return self.font_mappings.get(language_code, {}).get("name", language_code)
return self.font_mappings.get(key, {}).get("name", key)
def is_rtl(self, language_code):
"""
@@ -67,8 +90,10 @@ class FontConfig:
Raises:
ValueError: If the language code is not found in the mappings.
"""
if language_code not in self.font_mappings:
try:
key = self._resolve_mapping_key(language_code)
except ValueError:
# Preserve historical error message for tests/backward-compat
raise ValueError(f"Language code '{language_code}' is not supported.")
# Return RTL info if available, default to False
return self.font_mappings.get(language_code, {}).get("rtl", False)
return self.font_mappings.get(key, {}).get("rtl", False)
@@ -17,6 +17,7 @@ from co_op_translator.config.constants import (
SUPPORTED_NOTEBOOK_EXTENSIONS,
SUPPORTED_IMAGE_EXTENSIONS,
)
from co_op_translator.utils.common.lang_utils import normalize_language_code
logger = logging.getLogger(__name__)
@@ -436,16 +437,19 @@ class DirectoryManager:
rel_parts = ()
lang_code = None
if len(rel_parts) >= 2 and rel_parts[0] in self.language_codes:
lang_code = rel_parts[0]
# New format: base.hash.ext
path_hash_segment = parts[-2]
base_name = ".".join(parts[:-2])
# Accept alias language folder names by normalizing to canonical
if len(rel_parts) >= 2:
parent_lang = rel_parts[0]
normalized_parent = normalize_language_code(parent_lang)
if normalized_parent in self.language_codes:
lang_code = normalized_parent
path_hash_segment = parts[-2]
base_name = ".".join(parts[:-2])
else:
# Legacy format: base.hash.lang.ext
if len(parts) < 4:
continue
lang_code = parts[-2]
lang_code = normalize_language_code(parts[-2])
path_hash_segment = parts[-3]
base_name = ".".join(parts[:-3])
@@ -0,0 +1,143 @@
from __future__ import annotations
import logging
from dataclasses import dataclass
from pathlib import Path
from typing import List, Tuple
from co_op_translator.utils.common.lang_utils import ALIAS_TO_BCP47
logger = logging.getLogger(__name__)
@dataclass
class MigrationEntry:
base_dir: Path
source_dir: Path
dest_dir: Path
alias: str
canonical: str
conflict: bool = False
class LanguageFolderMigrator:
def __init__(
self,
root_dir: Path,
translations_dir: Path | None = None,
image_dir: Path | None = None,
):
self.root_dir = Path(root_dir)
self.translations_dir = (
(self.root_dir / "translations")
if translations_dir is None
else Path(translations_dir)
)
self.image_dir = (
(self.root_dir / "translated_images")
if image_dir is None
else Path(image_dir)
)
self.image_fast_dir = self.root_dir / "translated_images_fast"
self._aliases = set(ALIAS_TO_BCP47.keys())
def detect_alias_folders(self) -> List[MigrationEntry]:
entries: List[MigrationEntry] = []
for base in [self.translations_dir, self.image_dir, self.image_fast_dir]:
if not base.exists() or not base.is_dir():
continue
try:
for child in base.iterdir():
if not child.is_dir():
continue
alias = child.name
if alias in self._aliases:
canonical = ALIAS_TO_BCP47[alias]
dest = base / canonical
conflict = dest.exists() and any(dest.iterdir())
entries.append(
MigrationEntry(
base_dir=base,
source_dir=child,
dest_dir=dest,
alias=alias,
canonical=canonical,
conflict=conflict,
)
)
except Exception as e:
logger.debug(f"Failed scanning {base}: {e}")
return entries
@staticmethod
def format_plan(entries: List[MigrationEntry]) -> str:
if not entries:
return "No non-standard language folders detected."
lines: List[str] = ["Detected non-standard language folders:"]
for e in entries:
status = "(conflict)" if e.conflict else ""
lines.append(
f"- {e.source_dir.relative_to(e.base_dir)} -> {e.dest_dir.relative_to(e.base_dir)} {status}"
)
return "\n".join(lines)
def _fs_rename(self, src: Path, dst: Path) -> Tuple[bool, str]:
try:
dst.parent.mkdir(parents=True, exist_ok=True)
src.rename(dst)
return True, ""
except Exception as e:
return False, str(e)
def _rewrite_lang_metadata(self, lang_dir: Path, canonical_code: str) -> None:
"""Ensure per-language metadata under lang_dir stores canonical language_code.
Delegates to normalize_language_codes_in_lang_metadata to avoid duplication.
"""
try:
from co_op_translator.utils.common.metadata_utils import (
normalize_language_codes_in_lang_metadata,
)
normalize_language_codes_in_lang_metadata(lang_dir, canonical_code)
except Exception:
# Non-fatal if metadata normalization fails
pass
def execute(
self, entries: List[MigrationEntry], use_git: bool | None = None, dry_run: bool = False
) -> Tuple[int, List[str]]:
"""Execute migration entries. Skips conflicting destinations.
Returns number of successful renames and a list of messages for errors or conflicts.
Note: `use_git` is currently ignored and preserved only for backward compatibility.
"""
if not entries:
return 0, []
msgs: List[str] = []
successes = 0
for e in entries:
if e.conflict:
msgs.append(
f"Conflict: '{e.dest_dir}' already exists. Skipping auto-merge for '{e.source_dir}'."
)
continue
if dry_run:
# No changes, preview only
continue
ok, err = self._fs_rename(e.source_dir, e.dest_dir)
if ok:
successes += 1
# After successful rename, rewrite metadata to canonical code
try:
self._rewrite_lang_metadata(e.dest_dir, e.canonical)
except Exception:
# Non-fatal if metadata rewrite fails
pass
else:
msgs.append(
f"Failed to rename '{e.source_dir}' -> '{e.dest_dir}': {err}"
)
return successes, msgs
@@ -14,6 +14,11 @@ from co_op_translator.config.constants import (
SUPPORTED_IMAGE_EXTENSIONS,
SUPPORTED_NOTEBOOK_EXTENSIONS,
)
from co_op_translator.utils.common.lang_utils import (
normalize_language_codes,
get_supported_language_codes,
ALIAS_TO_BCP47,
)
from .directory_manager import DirectoryManager
from .translation_manager import TranslationManager
@@ -46,7 +51,8 @@ class ProjectTranslator:
root_dir: Root directory of the project to translate
translation_types: List of file types to translate (e.g., ["markdown", "images", "notebook"])
"""
self.language_codes = language_codes.split()
# Normalize to canonical BCP 47 (accept alias input like tw/cn/br)
self.language_codes = normalize_language_codes(language_codes.split())
self.root_dir = Path(root_dir).resolve()
# Resolve translations_dir relative to root_dir when a relative path is provided.
if translations_dir is not None:
@@ -83,6 +89,21 @@ class ProjectTranslator:
except ValueError:
# Directory is outside root_dir; no need to exclude from root scans
continue
# Dynamically add root-level language code folders (canonical or alias) to exclusions
try:
present_subdirs = [p for p in self.root_dir.iterdir() if p.is_dir()]
except Exception:
present_subdirs = []
supported = set(get_supported_language_codes())
aliases = set(ALIAS_TO_BCP47.keys())
for sub in present_subdirs:
name = sub.name
if name in supported or name in aliases:
# Only add absolute path strings to avoid substring false-positives in DirectoryManager
try:
excluded_dirs.add(str(sub.resolve()))
except Exception:
excluded_dirs.add(str(sub))
self.excluded_dirs = list(excluded_dirs)
# Initialize text translator
@@ -35,6 +35,9 @@ from co_op_translator.utils.common.task_utils import worker
from co_op_translator.utils.llm.markdown_utils import compare_line_breaks
from co_op_translator.utils.common.metadata_utils import is_notebook_up_to_date
from co_op_translator.config.base_config import Config
from co_op_translator.utils.common.file_utils import (
canonicalize_image_links_in_translations,
)
logger = logging.getLogger(__name__)
@@ -662,6 +665,20 @@ class TranslationManager:
migrated_md,
migrated_nb,
)
# As a safety net, canonicalize any remaining alias-based language dir segments in links
try:
md_fix, nb_fix = canonicalize_image_links_in_translations(
self.translations_dir, self.image_dir
)
if md_fix or nb_fix:
logger.info(
"Canonicalized image links in %d markdown and %d notebooks",
md_fix,
nb_fix,
)
except Exception as e:
logger.warning(f"Image link canonicalization skipped: {e}")
except Exception as e:
logger.warning(f"Image filename/link migration skipped: {e}")
@@ -25,16 +25,16 @@ ur:
name: "Urdu"
font: "NotoSansArabic-Medium.ttf"
rtl: true
zh:
zh-CN:
name: "Chinese (Simplified)"
font: "NotoSansCJK-Medium.ttc"
mo:
zh-MO:
name: "Chinese (Traditional, Macau)"
font: "NotoSansCJK-Medium.ttc"
hk:
zh-HK:
name: "Chinese (Traditional, Hong Kong)"
font: "NotoSansCJK-Medium.ttc"
tw:
zh-TW:
name: "Chinese (Traditional, Taiwan)"
font: "NotoSansCJK-Medium.ttc"
ja:
@@ -58,11 +58,11 @@ ne:
pa:
name: "Punjabi (Gurmukhi)"
font: "NotoSansGurmukhi-Medium.ttf"
pt:
pt-PT:
name: "Portuguese (Portugal)"
font: "NotoSans-Medium.ttf"
br:
pt-BR:
name: "Portuguese (Brazil)"
font: "NotoSans-Medium.ttf"
@@ -1,3 +1,3 @@
<!-- CO-OP TRANSLATOR LANGUAGES TABLE START -->
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh/README.md) | [Chinese (Traditional, Hong Kong)](./translations/hk/README.md) | [Chinese (Traditional, Macau)](./translations/mo/README.md) | [Chinese (Traditional, Taiwan)](./translations/tw/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Kannada](./translations/kn/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Malayalam](./translations/ml/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Nigerian Pidgin](./translations/pcm/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/br/README.md) | [Portuguese (Portugal)](./translations/pt/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Telugu](./translations/te/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh-CN/README.md) | [Chinese (Traditional, Hong Kong)](./translations/zh-HK/README.md) | [Chinese (Traditional, Macau)](./translations/zh-MO/README.md) | [Chinese (Traditional, Taiwan)](./translations/zh-TW/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Kannada](./translations/kn/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Malayalam](./translations/ml/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Nigerian Pidgin](./translations/pcm/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/pt-BR/README.md) | [Portuguese (Portugal)](./translations/pt-PT/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Telugu](./translations/te/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
<!-- CO-OP TRANSLATOR LANGUAGES TABLE END -->
+140 -11
View File
@@ -423,15 +423,49 @@ def filter_files(directory: str | Path, excluded_dirs, extension: str = None) ->
directory = Path(directory)
files = []
# Normalize excluded directories into two buckets: absolute paths and names
abs_excluded: list[Path] = []
name_excluded: set[str] = set()
for item in excluded_dirs:
try:
p = Path(item)
if p.is_absolute():
abs_excluded.append(p.resolve())
else:
name_excluded.add(str(item))
except Exception:
name_excluded.add(str(item))
# Recursively traverse the directory
for path in directory.rglob("*"):
# Check if the path is a file, matches extension if specified, and doesn't contain excluded dirs
if (
path.is_file()
and (extension is None or path.suffix.lower() == extension.lower())
and not any(excluded_dir in path.parts for excluded_dir in excluded_dirs)
):
files.append(path)
if not path.is_file():
continue
if extension is not None and path.suffix.lower() != extension.lower():
continue
# Exclude by absolute path ancestry
excluded_by_abs = False
try:
resolved = path.resolve()
for abs_dir in abs_excluded:
try:
resolved.relative_to(abs_dir)
excluded_by_abs = True
break
except Exception:
continue
except Exception:
# If resolve() fails, fall back to name-based exclusion only
excluded_by_abs = False
if excluded_by_abs:
continue
# Name-based exclusion (segment match)
if any(ex in path.parts for ex in name_excluded):
continue
files.append(path)
return files
@@ -462,6 +496,8 @@ def migrate_translated_image_filenames(
except Exception:
return {}
from co_op_translator.utils.common.lang_utils import normalize_language_code
for image_file in image_files:
if not image_file.is_file():
continue
@@ -486,18 +522,26 @@ def migrate_translated_image_filenames(
extension = parts[-1]
# Detect language either from directory or from legacy filename
under_lang_dir = len(rel_parts) >= 2 and rel_parts[0] in language_codes
lang_code: str | None = rel_parts[0] if under_lang_dir else None
under_lang_dir = False
lang_code: str | None = None
if len(rel_parts) >= 2:
parent_lang = rel_parts[0]
normalized_parent = normalize_language_code(parent_lang)
if normalized_parent in language_codes:
under_lang_dir = True
lang_code = normalized_parent
# Legacy filename pattern includes trailing language segment
has_legacy_lang_in_name = len(parts) >= 4 and parts[-2] in language_codes
has_legacy_lang_in_name = (
len(parts) >= 4 and normalize_language_code(parts[-2]) in language_codes
)
if not under_lang_dir and not has_legacy_lang_in_name:
# Cannot determine language; skip conservatively
continue
if has_legacy_lang_in_name and lang_code is None:
lang_code = parts[-2]
lang_code = normalize_language_code(parts[-2])
if lang_code not in language_codes:
# Skip unsupported languages
@@ -801,3 +845,88 @@ def delete_translated_markdown_files_by_language_code(
# Remove the entire directory and its contents
shutil.rmtree(language_dir)
logger.info(f"Deleted the directory and all files for language: {language_code}")
def canonicalize_image_links_in_translations(
translations_dir: Path, image_dir: Path
) -> tuple[int, int]:
"""
Canonicalize image links in translated markdown and notebooks by rewriting
alias-based language directory segments to canonical BCP 47.
Examples:
translated_images/tw/... -> translated_images/zh-TW/...
translated_images/cn/... -> translated_images/zh-CN/...
<base_dir>/br/... -> <base_dir>/pt-BR/...
The function scans under translations_dir and updates files in-place.
Returns:
(md_files_updated, nb_files_updated)
"""
from co_op_translator.utils.common.lang_utils import ALIAS_TO_BCP47
from co_op_translator.config.constants import (
SUPPORTED_MARKDOWN_EXTENSIONS,
SUPPORTED_NOTEBOOK_EXTENSIONS,
)
translations_dir = Path(translations_dir)
image_dir = Path(image_dir)
base_dir_name = image_dir.name
base_dirs = [base_dir_name, "translated_images", "translated_images_fast"]
def _canonicalize_text(text: str) -> str:
updated = text
for bdir in base_dirs:
for alias, canonical in ALIAS_TO_BCP47.items():
updated = updated.replace(f"{bdir}/{alias}/", f"{bdir}/{canonical}/")
# Also replace Windows-style separators just in case
updated = updated.replace(
f"{bdir}\\{alias}\\", f"{bdir}\\{canonical}\\"
)
return updated
md_updated = 0
nb_updated = 0
# Markdown files
try:
md_files: list[Path] = []
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
md_files.extend(translations_dir.rglob(f"*{ext}"))
for md in md_files:
try:
original = md.read_text(encoding="utf-8")
except Exception:
continue
updated = _canonicalize_text(original)
if updated != original:
try:
md.write_text(updated, encoding="utf-8")
md_updated += 1
except Exception:
pass
except Exception:
pass
# Notebooks (JSON)
try:
nb_files: list[Path] = []
for ext in SUPPORTED_NOTEBOOK_EXTENSIONS:
nb_files.extend(translations_dir.rglob(f"*{ext}"))
for nb in nb_files:
try:
content = nb.read_text(encoding="utf-8")
except Exception:
continue
updated = _canonicalize_text(content)
if updated != content:
try:
nb.write_text(updated, encoding="utf-8")
nb_updated += 1
except Exception:
pass
except Exception:
pass
return md_updated, nb_updated
@@ -0,0 +1,109 @@
from __future__ import annotations
import importlib.resources
import logging
import re
from pathlib import Path
from typing import Iterable, List, Tuple
import yaml
logger = logging.getLogger(__name__)
# Explicit alias map only (no heuristics). All keys are lower-case.
ALIAS_TO_BCP47: dict[str, str] = {
# Chinese regions
"cn": "zh-CN",
"tw": "zh-TW",
"hk": "zh-HK",
"mo": "zh-MO",
# Portuguese regions
"br": "pt-BR",
"pt": "pt-PT", # treat bare 'pt' as Portugal when normalizing
# Common country-code aliases
"jp": "ja",
"kr": "ko",
# Generic zh alias maps to Mainland China for our purposes
"zh": "zh-CN",
}
# BCP 47 basic pattern: language[-script][-region][-variants...]
# We only standardize language + region casing here, not validating scripts/variants.
_BCP47_SPLIT = re.compile(r"[-_]")
def canonical_case(code: str) -> str:
"""Apply canonical casing: language lower-case, region upper-case.
Examples:
- "pt-br" -> "pt-BR"
- "ZH-tw" -> "zh-TW"
- "ja" -> "ja"
"""
parts = _BCP47_SPLIT.split(code.strip())
if not parts:
return code
lang = parts[0].lower()
rest: List[str] = []
for i, p in enumerate(parts[1:], start=1):
if len(p) == 2: # region
rest.append(p.upper())
else:
# leave script/variants as-is but prefer title for script (4 letters)
if len(p) == 4:
rest.append(p.title())
else:
rest.append(p)
return "-".join([lang, *rest]) if rest else lang
def normalize_language_code(code: str) -> str:
"""Normalize an input code to our canonical BCP 47 form using explicit aliases.
- Applies explicit alias mapping first (exact match on lower-cased input)
- Then applies canonical casing rules.
"""
raw = code.strip()
if not raw:
return raw
key = raw.lower()
if key in ALIAS_TO_BCP47:
return ALIAS_TO_BCP47[key]
return canonical_case(raw)
def normalize_language_codes(codes: Iterable[str]) -> List[str]:
seen: set[str] = set()
normalized: List[str] = []
for c in codes:
canon = normalize_language_code(c)
if canon not in seen and canon:
seen.add(canon)
normalized.append(canon)
return normalized
def get_supported_language_codes() -> List[str]:
"""Return canonical supported codes (keys) from font_language_mappings.yml.
Note: This list reflects our packaging and is used when "all" is selected.
"""
try:
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file) or {}
return [
lang_code
for lang_code, meta in font_mappings.items()
if isinstance(meta, dict)
]
except Exception as e:
logger.warning(f"Failed to load font mappings: {e}")
return []
def is_supported_language(code: str) -> bool:
canon = normalize_language_code(code)
return canon in set(get_supported_language_codes())
@@ -671,3 +671,35 @@ def remove_text_metadata_for_source(lang_dir: Path, source_file: str | Path) ->
pass
if changed:
_save_lang_metadata(lang_dir, all_metadata)
def normalize_language_codes_in_lang_metadata(
lang_dir: Path, canonical_code: str
) -> int:
"""Normalize 'language_code' fields in .co-op-translator.json under a language folder.
Args:
lang_dir: Path to translations/<lang> or translated_images/<lang> directory
canonical_code: Expected canonical BCP47 language code (e.g., 'pt-BR')
Returns:
Number of entries updated.
"""
lang_dir = Path(lang_dir)
metadata_path = _get_metadata_file_path(lang_dir)
if not metadata_path.exists():
return 0
try:
data = _load_lang_metadata(lang_dir)
if not isinstance(data, dict) or not data:
return 0
changed = 0
for k, v in list(data.items()):
if isinstance(v, dict) and v.get("language_code") != canonical_code:
v["language_code"] = canonical_code
changed += 1
if changed:
_save_lang_metadata(lang_dir, data)
return changed
except Exception:
return 0
@@ -0,0 +1,62 @@
import pytest
from pathlib import Path
from unittest.mock import patch, mock_open
from co_op_translator.config.font_config import FontConfig
sample_yaml = """
zh-TW:
name: Chinese (Traditional, Taiwan)
font: "NotoSansCJK-Medium.ttc"
pt-PT:
name: Portuguese (Portugal)
font: "NotoSans-Medium.ttf"
pt-BR:
name: Portuguese (Brazil)
font: "NotoSans-Medium.ttf"
"""
def test_font_config_resolves_canonical_to_alias_keys():
# Mock YAML with alias keys only
with (
patch("importlib.resources.path") as mock_path_yaml,
patch("builtins.open", mock_open(read_data=sample_yaml)),
):
mock_path_yaml.return_value = Path("fake/fonts/font_language_mappings.yml")
fc = FontConfig()
# get_font_path should resolve alias input 'tw' to canonical 'zh-TW'
with patch(
"importlib.resources.path",
return_value=Path("fake_fonts/NotoSansCJK-Medium.ttc"),
) as mock_path_font:
path = fc.get_font_path("tw")
assert Path(path).name == "NotoSansCJK-Medium.ttc"
mock_path_font.assert_called_once_with(
"co_op_translator.fonts", "NotoSansCJK-Medium.ttc"
)
# get_language_name should resolve alias input 'br' to canonical 'pt-BR'
name = fc.get_language_name("br")
assert name == "Portuguese (Brazil)"
# is_rtl defaults to False if not set
assert fc.is_rtl("zh-TW") is False
def test_font_config_invalid_language_errors():
with (
patch("importlib.resources.path") as mock_path_yaml,
patch("builtins.open", mock_open(read_data=sample_yaml)),
):
mock_path_yaml.return_value = Path("fake/fonts/font_language_mappings.yml")
fc = FontConfig()
with pytest.raises(ValueError) as excinfo:
fc.get_language_name("xx")
assert "Language code 'xx' is not supported." in str(excinfo.value)
with pytest.raises(ValueError) as excinfo2:
fc.is_rtl("xx")
assert "Language code 'xx' is not supported." in str(excinfo2.value)
@@ -118,7 +118,7 @@ class TestDirectoryManager:
# Create translated image with hash under image_dir (default: root_dir/translated_images)
translations_dir.mkdir(exist_ok=True)
image_dir = root_dir / "translated_images"
ko_img_dir = image_dir / "img"
ko_img_dir = image_dir / "ko" / "img"
ko_img_dir.mkdir(parents=True, exist_ok=True)
# Valid translated image (with correct hash)
@@ -126,12 +126,13 @@ class TestDirectoryManager:
original_name, ext = os.path.splitext(original_img.name)
path_hash = get_unique_id(original_img, root_dir)
valid_trans_name = f"{original_name}.{path_hash}.ko{ext}"
# Canonical layout: translated_images/<lang>/<basename>.<hash>.<ext>
valid_trans_name = f"{original_name}.{path_hash}{ext}"
valid_trans_img = ko_img_dir / valid_trans_name
valid_trans_img.write_bytes(b"translated content")
# Orphaned translated image (with incorrect hash)
orphaned_trans_img = ko_img_dir / "test.invalid_hash.ko.png"
orphaned_trans_img = ko_img_dir / "test.invalid_hash.png"
orphaned_trans_img.write_bytes(b"orphaned content")
manager = DirectoryManager(
@@ -0,0 +1,31 @@
from pathlib import Path
from co_op_translator.core.project.project_translator import ProjectTranslator
from co_op_translator.utils.common.file_utils import filter_files
def test_dynamic_exclusion_of_root_language_dirs(tmp_path: Path):
root = tmp_path
# Create root-level language dirs (canonical and alias) and a normal docs folder
(root / "zh-TW").mkdir()
(root / "cn").mkdir()
(root / "docs").mkdir()
# Create files
(root / "zh-TW" / "a.md").write_text("x", encoding="utf-8")
(root / "cn" / "b.md").write_text("y", encoding="utf-8")
(root / "docs" / "c.md").write_text("z", encoding="utf-8")
# Initialize translator (this will compute dynamic exclusions)
pt = ProjectTranslator(
language_codes="ko",
root_dir=str(root),
translation_types=["markdown"],
)
files = filter_files(root, pt.excluded_dirs, extension=".md")
paths = {str(p.relative_to(root)).replace("\\", "/") for p in files}
# Only docs/c.md should be present; language-dir files must be excluded
assert "docs/c.md" in paths
assert "zh-TW/a.md" not in paths
assert "cn/b.md" not in paths
@@ -0,0 +1,55 @@
from pathlib import Path
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
def test_detect_alias_folders(tmp_path: Path):
# Arrange a fake project with alias folders
root = tmp_path
translations = root / "translations"
images = root / "translated_images"
fast = root / "translated_images_fast"
(translations / "tw").mkdir(parents=True)
(images / "cn").mkdir(parents=True)
(fast / "br").mkdir(parents=True)
migrator = LanguageFolderMigrator(root_dir=root)
# Act
entries = migrator.detect_alias_folders()
# Assert
mapping = {(e.base_dir.name, e.alias, e.canonical) for e in entries}
assert ("translations", "tw", "zh-TW") in mapping
assert ("translated_images", "cn", "zh-CN") in mapping
assert ("translated_images_fast", "br", "pt-BR") in mapping
def test_execute_fs_rename(tmp_path: Path):
# Arrange alias source and missing canonical destination
root = tmp_path
translations = root / "translations"
(translations / "tw").mkdir(parents=True)
(translations / "tw" / "foo").mkdir()
migrator = LanguageFolderMigrator(root_dir=root)
entries = migrator.detect_alias_folders()
# Precondition
assert (translations / "tw").exists()
assert not (translations / "zh-TW").exists()
# Act: execute with dry_run first
renamed, msgs = migrator.execute(entries, use_git=False, dry_run=True)
# No changes in dry run
assert renamed == 0
assert (translations / "tw").exists()
assert not (translations / "zh-TW").exists()
# Act: real rename (filesystem)
renamed, msgs = migrator.execute(entries, use_git=False, dry_run=False)
# Assert rename happened
assert renamed == 1
assert not (translations / "tw").exists()
assert (translations / "zh-TW").exists()
@@ -0,0 +1,113 @@
import json
from pathlib import Path
from co_op_translator.utils.common.file_utils import (
canonicalize_image_links_in_translations,
)
def write_text(p: Path, content: str) -> None:
p.parent.mkdir(parents=True, exist_ok=True)
p.write_text(content, encoding="utf-8")
def test_markdown_rewrites_aliases_for_multiple_bases(tmp_path: Path):
translations_dir = tmp_path / "translations"
image_dir = tmp_path / "images_out" # custom base dir name
# Create a markdown file under translations with multiple alias patterns
md = translations_dir / "ko" / "docs" / "sample.md"
original = (
"![t1](translated_images/tw/foo.webp)\n"
"![t2](translated_images_fast/cn/bar.webp)\n"
f"![t3]({image_dir.name}/br/baz.webp)\n"
)
write_text(md, original)
md_updated, nb_updated = canonicalize_image_links_in_translations(
translations_dir=translations_dir, image_dir=image_dir
)
# One markdown file updated, no notebooks
assert md_updated == 1
assert nb_updated == 0
updated = md.read_text(encoding="utf-8")
assert "translated_images/zh-TW/foo.webp" in updated
assert "translated_images_fast/zh-CN/bar.webp" in updated
assert f"{image_dir.name}/pt-BR/baz.webp" in updated
def test_notebook_rewrites_aliases(tmp_path: Path):
translations_dir = tmp_path / "translations"
image_dir = tmp_path / "translated_images" # common default base
nb = translations_dir / "ja" / "nb" / "sample.ipynb"
nb_content = {
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"![x](translated_images/cn/a.webp)\n",
"text before ",
"![y](translated_images_fast/br/b.webp)\n",
],
}
],
"metadata": {},
"nbformat": 4,
"nbformat_minor": 5,
}
nb.parent.mkdir(parents=True, exist_ok=True)
nb.write_text(json.dumps(nb_content), encoding="utf-8")
md_updated, nb_updated = canonicalize_image_links_in_translations(
translations_dir=translations_dir, image_dir=image_dir
)
assert md_updated == 0
assert nb_updated == 1
updated = nb.read_text(encoding="utf-8")
assert "translated_images/zh-CN/a.webp" in updated
assert "translated_images_fast/pt-BR/b.webp" in updated
def test_no_changes_returns_zero(tmp_path: Path):
translations_dir = tmp_path / "translations"
image_dir = tmp_path / "images_out"
md = translations_dir / "fr" / "doc.md"
write_text(md, "No image paths here.")
md_updated, nb_updated = canonicalize_image_links_in_translations(
translations_dir=translations_dir, image_dir=image_dir
)
assert (md_updated, nb_updated) == (0, 0)
def test_windows_separators_are_handled(tmp_path: Path):
translations_dir = tmp_path / "translations"
image_dir = tmp_path / "images_out"
md = translations_dir / "de" / "doc.md"
original = (
"![t1](translated_images\\tw\\foo.webp)\n"
"![t2](translated_images_fast\\br\\bar.webp)\n"
f"![t3]({image_dir.name}\\cn\\baz.webp)\n"
)
write_text(md, original)
md_updated, nb_updated = canonicalize_image_links_in_translations(
translations_dir=translations_dir, image_dir=image_dir
)
assert md_updated == 1
assert nb_updated == 0
updated = md.read_text(encoding="utf-8")
assert "translated_images\\zh-TW\\foo.webp" in updated
assert "translated_images_fast\\pt-BR\\bar.webp" in updated
assert f"{image_dir.name}\\zh-CN\\baz.webp" in updated
@@ -0,0 +1,38 @@
import pytest
from co_op_translator.utils.common.lang_utils import (
normalize_language_code,
normalize_language_codes,
ALIAS_TO_BCP47,
)
def test_normalize_language_code_aliases():
assert normalize_language_code("tw") == "zh-TW"
assert normalize_language_code("cn") == "zh-CN"
assert normalize_language_code("br") == "pt-BR"
assert normalize_language_code("jp") == "ja"
assert normalize_language_code("kr") == "ko"
# generic zh maps to zh-CN per policy
assert normalize_language_code("zh") == "zh-CN"
def test_normalize_language_code_casing():
assert normalize_language_code("pt-br") == "pt-BR"
assert normalize_language_code("PT-br") == "pt-BR"
assert normalize_language_code("zh-tw") == "zh-TW"
assert normalize_language_code("EN-us") == "en-US"
assert normalize_language_code("ja") == "ja"
def test_normalize_language_codes_dedup_order():
items = ["tw", "zh-TW", "cn", "ZH-cn", "br", "pt-br", "jp", "kr"]
# Expect deduped canonical order of first occurrences
out = normalize_language_codes(items)
assert out == ["zh-TW", "zh-CN", "pt-BR", "ja", "ko"]
def test_alias_table_contains_expected():
# minimal sanity of important aliases
for k in ["tw", "cn", "br", "jp", "kr", "zh"]:
assert k in ALIAS_TO_BCP47