mirror of
https://github.com/Azure/co-op-translator
synced 2026-08-09 12:00:08 +00:00
Core: Standardize language codes to canonical BCP 47 across CLI, core, templates, and metadata (#350)
This commit is contained in:
@@ -1,9 +1,9 @@
|
||||
# 🌐 Multi-Language Support (Template)
|
||||
|
||||
Maintainers: The block below is an "all languages" example managed by Co‑op Translator.
|
||||
Maintainers: The block below is an "all languages" example managed by Co-op Translator.
|
||||
|
||||
- If you want Co‑op Translator to keep this list fully up‑to‑date automatically when you run `translate` (any language selection), keep the two comment markers as-is.
|
||||
- If you only want to show a subset of languages, delete the two comment markers and remove any languages you don't want to list. After removing the markers, Co‑op Translator will no longer auto‑replace this section.
|
||||
- If you want Co-op Translator to keep this list fully up-to-date automatically when you run `translate` (any language selection), keep the two comment markers as-is.
|
||||
- If you only want to show a subset of languages, delete the two comment markers and remove any languages you don't want to list. After removing the markers, Co-op Translator will no longer auto-replace this section.
|
||||
|
||||
- The section now includes a "Prefer to Clone Locally?" advisory to help users clone without the large translations payload. You can personalize the advisory with your repository URL by running, for example:
|
||||
- `translate -l "ko" --repo-url "https://github.com/org/repo.git"`
|
||||
@@ -15,7 +15,7 @@ Maintainers: The block below is an "all languages" example managed by Co‑op Tr
|
||||
#### Supported by [Co-op Translator](https://github.com/Azure/Co-op-Translator)
|
||||
|
||||
<!-- CO-OP TRANSLATOR LANGUAGES TABLE START -->
|
||||
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh/README.md) | [Chinese (Traditional, Hong Kong)](./translations/hk/README.md) | [Chinese (Traditional, Macau)](./translations/mo/README.md) | [Chinese (Traditional, Taiwan)](./translations/tw/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/br/README.md) | [Portuguese (Portugal)](./translations/pt/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
|
||||
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh-CN/README.md) | [Chinese (Traditional, Hong Kong)](./translations/zh-HK/README.md) | [Chinese (Traditional, Macau)](./translations/zh-MO/README.md) | [Chinese (Traditional, Taiwan)](./translations/zh-TW/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/pt-BR/README.md) | [Portuguese (Portugal)](./translations/pt-PT/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
|
||||
> **Prefer to Clone Locally?**
|
||||
>
|
||||
> This repository includes 50+ language translations which significantly increases the download size. To clone without translations, use sparse checkout:
|
||||
@@ -29,3 +29,4 @@ Maintainers: The block below is an "all languages" example managed by Co‑op Tr
|
||||
<!-- CO-OP TRANSLATOR LANGUAGES TABLE END -->
|
||||
|
||||
```
|
||||
|
||||
|
||||
@@ -12,10 +12,10 @@ The table below lists the languages currently supported by **Co-op Translator**.
|
||||
| ar | Arabic | NotoSansArabic-Medium.ttf | Yes | No |
|
||||
| fa | Persian (Farsi) | NotoSansArabic-Medium.ttf | Yes | No |
|
||||
| ur | Urdu | NotoSansArabic-Medium.ttf | Yes | No |
|
||||
| zh | Chinese (Simplified) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| mo | Chinese (Traditional, Macau) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| hk | Chinese (Traditional, Hong Kong) | NotoSansCJK-Medium.ttc| No | No |
|
||||
| tw | Chinese (Traditional, Taiwan) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| zh-CN | Chinese (Simplified) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| zh-MO | Chinese (Traditional, Macau) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| zh-HK | Chinese (Traditional, Hong Kong) | NotoSansCJK-Medium.ttc| No | No |
|
||||
| zh-TW | Chinese (Traditional, Taiwan) | NotoSansCJK-Medium.ttc | No | No |
|
||||
| ja | Japanese | NotoSansCJK-Medium.ttc | No | No |
|
||||
| ko | Korean | NotoSansCJK-Medium.ttc | No | No |
|
||||
| hi | Hindi | NotoSansDevanagari-Medium.ttf | No | No |
|
||||
@@ -23,8 +23,8 @@ The table below lists the languages currently supported by **Co-op Translator**.
|
||||
| mr | Marathi | NotoSansDevanagari-Medium.ttf | No | No |
|
||||
| ne | Nepali | NotoSansDevanagari-Medium.ttf | No | No |
|
||||
| pa | Punjabi (Gurmukhi) | NotoSansGurmukhi-Medium.ttf | No | No |
|
||||
| pt | Portuguese (Portugal)| NotoSans-Medium.ttf | No | No |
|
||||
| br | Portuguese (Brazil) | NotoSans-Medium.ttf | No | No |
|
||||
| pt-PT | Portuguese (Portugal)| NotoSans-Medium.ttf | No | No |
|
||||
| pt-BR | Portuguese (Brazil) | NotoSans-Medium.ttf | No | No |
|
||||
| it | Italian | NotoSans-Medium.ttf | No | No |
|
||||
| lt | Lithuanian | NotoSans-Medium.ttf | No | No |
|
||||
| pl | Polish | NotoSans-Medium.ttf | No | No |
|
||||
|
||||
@@ -11,6 +11,7 @@ import os
|
||||
from co_op_translator.core.project.project_evaluator import ProjectEvaluator
|
||||
from co_op_translator.config.base_config import Config
|
||||
from co_op_translator.utils.common.logging_utils import setup_logging
|
||||
from co_op_translator.utils.common.lang_utils import normalize_language_code
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -93,7 +94,9 @@ def evaluate_command(
|
||||
if save_logs and log_file_path is not None:
|
||||
click.echo(f"📄 Logs will be saved to: {log_file_path}")
|
||||
|
||||
click.echo(f"Evaluating {language_code} translations in {root_path}...")
|
||||
# Normalize to canonical BCP 47 (accept alias input like tw/cn/br)
|
||||
canonical_code = normalize_language_code(language_code)
|
||||
click.echo(f"Evaluating {canonical_code} translations in {root_path}...")
|
||||
|
||||
# Create evaluator
|
||||
# Determine evaluation mode (fast, deep or default mode)
|
||||
@@ -121,7 +124,7 @@ def evaluate_command(
|
||||
evaluator = ProjectEvaluator(
|
||||
root_dir=root_path,
|
||||
translations_dir=root_path / "translations",
|
||||
language_codes=[language_code],
|
||||
language_codes=[canonical_code],
|
||||
excluded_dirs=["node_modules", ".git", "__pycache__", "venv"],
|
||||
use_llm=use_llm,
|
||||
use_rule=use_rule,
|
||||
@@ -129,7 +132,7 @@ def evaluate_command(
|
||||
|
||||
# Run evaluation
|
||||
total_files, issue_files, avg_confidence = asyncio.run(
|
||||
evaluator.evaluate_project(language_code)
|
||||
evaluator.evaluate_project(canonical_code)
|
||||
)
|
||||
|
||||
# Display results with color highlighting
|
||||
@@ -154,7 +157,7 @@ def evaluate_command(
|
||||
|
||||
# Get low confidence translations
|
||||
low_confidence = asyncio.run(
|
||||
evaluator.get_low_confidence_translations(language_code, min_confidence)
|
||||
evaluator.get_low_confidence_translations(canonical_code, min_confidence)
|
||||
)
|
||||
|
||||
if low_confidence:
|
||||
@@ -175,10 +178,10 @@ def evaluate_command(
|
||||
|
||||
# Check if the path already contains translations/language_code
|
||||
# to avoid duplication
|
||||
if rel_path.startswith(f"translations/{language_code}/"):
|
||||
if rel_path.startswith(f"translations/{canonical_code}/"):
|
||||
display_path = f"./{rel_path}"
|
||||
else:
|
||||
display_path = f"./translations/{language_code}/{rel_path}"
|
||||
display_path = f"./translations/{canonical_code}/{rel_path}"
|
||||
except ValueError:
|
||||
display_path = str(file_path).replace("\\", "/")
|
||||
|
||||
@@ -190,7 +193,7 @@ def evaluate_command(
|
||||
)
|
||||
|
||||
trans_path = Path(file_path)
|
||||
lang_dir = root_path / "translations" / language_code
|
||||
lang_dir = root_path / "translations" / canonical_code
|
||||
try:
|
||||
rel = trans_path.resolve().relative_to(lang_dir)
|
||||
orig_path = root_path / rel
|
||||
@@ -239,7 +242,7 @@ def evaluate_command(
|
||||
f"\n{click.style('Note:', fg='yellow')} Files with issues were found during evaluation, but none fall below the confidence threshold of {min_confidence}."
|
||||
)
|
||||
click.echo(
|
||||
f"Consider running with a higher threshold: {click.style(f'evaluate -l {language_code} --min-confidence 0.9', bold=True)}"
|
||||
f"Consider running with a higher threshold: {click.style(f'evaluate -l {canonical_code} --min-confidence 0.9', bold=True)}"
|
||||
)
|
||||
|
||||
else:
|
||||
@@ -247,7 +250,7 @@ def evaluate_command(
|
||||
f"\n{click.style('✓ All translations look good!', fg='green', bold=True)}"
|
||||
)
|
||||
|
||||
logger.info(f"Evaluation completed for language: {language_code}")
|
||||
logger.info(f"Evaluation completed for language: {canonical_code}")
|
||||
|
||||
except Exception as e:
|
||||
if debug:
|
||||
|
||||
@@ -13,20 +13,22 @@ import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
from tqdm import tqdm
|
||||
import importlib.resources
|
||||
import yaml
|
||||
|
||||
from co_op_translator.config.base_config import Config
|
||||
from co_op_translator.utils.llm.markdown_utils import (
|
||||
migrate_notebook_links,
|
||||
update_notebook_links,
|
||||
)
|
||||
from co_op_translator.utils.common.file_utils import map_original_to_translated
|
||||
from co_op_translator.utils.common.file_utils import (
|
||||
map_original_to_translated,
|
||||
canonicalize_image_links_in_translations,
|
||||
)
|
||||
from co_op_translator.utils.common.logging_utils import setup_logging
|
||||
from co_op_translator.config.constants import (
|
||||
SUPPORTED_MARKDOWN_EXTENSIONS,
|
||||
SUPPORTED_NOTEBOOK_EXTENSIONS,
|
||||
)
|
||||
from co_op_translator.utils.common.lang_utils import normalize_language_codes
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -44,6 +46,11 @@ logger = logging.getLogger(__name__)
|
||||
default=".",
|
||||
help="Root directory of the project (default is current directory).",
|
||||
)
|
||||
@click.option(
|
||||
"--image-dir",
|
||||
default="translated_images",
|
||||
help="Base directory for translated images (relative to --root-dir).",
|
||||
)
|
||||
@click.option(
|
||||
"--dry-run",
|
||||
is_flag=True,
|
||||
@@ -74,6 +81,7 @@ logger = logging.getLogger(__name__)
|
||||
def migrate_links_command(
|
||||
language_codes,
|
||||
root_dir,
|
||||
image_dir,
|
||||
dry_run,
|
||||
fallback_to_original,
|
||||
debug,
|
||||
@@ -106,6 +114,19 @@ def migrate_links_command(
|
||||
click.echo(f"No translations directory found at: {translations_dir}")
|
||||
return
|
||||
|
||||
# Canonicalize legacy alias-based language segments in links across translated content
|
||||
try:
|
||||
md_fix, nb_fix = canonicalize_image_links_in_translations(
|
||||
translations_dir=translations_dir,
|
||||
image_dir=(root_path / image_dir),
|
||||
)
|
||||
if md_fix or nb_fix:
|
||||
click.echo(
|
||||
f"✅ Canonicalized alias language segments in links: markdown={md_fix}, notebooks={nb_fix}"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"Canonicalization step skipped: {e}")
|
||||
|
||||
# Warning and confirmation when processing all languages
|
||||
if isinstance(language_codes, str) and language_codes.lower() == "all":
|
||||
click.echo(
|
||||
@@ -129,34 +150,12 @@ def migrate_links_command(
|
||||
|
||||
# Parse language codes list (support "all")
|
||||
if isinstance(language_codes, str) and language_codes.lower() == "all":
|
||||
try:
|
||||
with importlib.resources.path(
|
||||
"co_op_translator.fonts", "font_language_mappings.yml"
|
||||
) as mappings_path:
|
||||
with open(mappings_path, "r", encoding="utf-8") as file:
|
||||
font_mappings = yaml.safe_load(file)
|
||||
if not font_mappings:
|
||||
raise click.ClickException("Empty font mappings file")
|
||||
lang_list = [
|
||||
lang_code
|
||||
for lang_code in font_mappings
|
||||
if isinstance(font_mappings[lang_code], dict)
|
||||
]
|
||||
if not lang_list:
|
||||
raise click.ClickException(
|
||||
"No valid language codes found in font mappings"
|
||||
)
|
||||
logging.debug(
|
||||
f"Expanded 'all' to language codes from font mapping: {lang_list}"
|
||||
)
|
||||
except (FileNotFoundError, yaml.YAMLError) as e:
|
||||
raise click.ClickException(
|
||||
f"Failed to load language codes for 'all': {str(e)}"
|
||||
)
|
||||
lang_list = Config.get_language_codes()
|
||||
else:
|
||||
lang_list = [
|
||||
code.strip() for code in language_codes.split() if code.strip()
|
||||
]
|
||||
lang_list = normalize_language_codes(lang_list)
|
||||
if not lang_list:
|
||||
raise click.ClickException("No valid language codes provided.")
|
||||
|
||||
@@ -181,7 +180,9 @@ def migrate_links_command(
|
||||
md_files: list[Path] = []
|
||||
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
|
||||
md_files.extend(lang_dir.rglob(f"*{ext}"))
|
||||
for md_translated in tqdm(md_files, desc=f"{lang} md", unit="file"):
|
||||
for md_translated in tqdm(
|
||||
md_files, desc=f"{lang_dir.name} md", unit="file"
|
||||
):
|
||||
total_scanned += 1
|
||||
try:
|
||||
content = md_translated.read_text(encoding="utf-8")
|
||||
@@ -260,7 +261,7 @@ def migrate_links_command(
|
||||
# Determine translated counterpart if exists
|
||||
candidate_translated = map_original_to_translated(
|
||||
original_abs=linked_abs,
|
||||
language_code=lang,
|
||||
language_code=lang_dir.name,
|
||||
root_dir=root_path,
|
||||
)
|
||||
|
||||
@@ -278,7 +279,7 @@ def migrate_links_command(
|
||||
# Compute translated_md_dir for later comparisons
|
||||
translated_md_dir = (
|
||||
translations_dir
|
||||
/ lang
|
||||
/ lang_dir.name
|
||||
/ original_md_path.relative_to(root_path).parent
|
||||
)
|
||||
|
||||
@@ -329,7 +330,7 @@ def migrate_links_command(
|
||||
updated = update_notebook_links(
|
||||
markdown_string=content,
|
||||
md_file_path=original_md_path,
|
||||
language_code=lang,
|
||||
language_code=lang_dir.name,
|
||||
translations_dir=translations_dir,
|
||||
root_dir=root_path,
|
||||
use_translated_notebook=True,
|
||||
@@ -338,7 +339,7 @@ def migrate_links_command(
|
||||
updated = migrate_notebook_links(
|
||||
markdown_string=content,
|
||||
md_file_path=original_md_path,
|
||||
language_code=lang,
|
||||
language_code=lang_dir.name,
|
||||
root_dir=root_path,
|
||||
)
|
||||
|
||||
|
||||
@@ -5,9 +5,6 @@ Translate command implementation for Co-op Translator CLI.
|
||||
import asyncio
|
||||
import logging
|
||||
import click
|
||||
import importlib.resources
|
||||
import yaml
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from co_op_translator.core.project.project_translator import ProjectTranslator
|
||||
@@ -19,6 +16,13 @@ from co_op_translator.utils.common.file_utils import (
|
||||
update_readme_languages_table,
|
||||
update_readme_other_courses,
|
||||
)
|
||||
from co_op_translator.utils.common.lang_utils import (
|
||||
normalize_language_codes,
|
||||
)
|
||||
from co_op_translator.utils.common.metadata_utils import (
|
||||
normalize_language_codes_in_lang_metadata,
|
||||
)
|
||||
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -91,6 +95,19 @@ logger = logging.getLogger(__name__)
|
||||
default=None,
|
||||
help="Repository URL to show in the 'Prefer to Clone Locally?' advisory inside the languages table.",
|
||||
)
|
||||
@click.option(
|
||||
"--migrate-language-folders",
|
||||
is_flag=True,
|
||||
help=(
|
||||
"Detect and optionally rename alias-based language folders (e.g., tw, cn, br) "
|
||||
"to canonical BCP 47 (zh-TW, zh-CN, pt-BR)."
|
||||
),
|
||||
)
|
||||
@click.option(
|
||||
"--dry-run",
|
||||
is_flag=True,
|
||||
help="Preview migration plan without making changes (use with --migrate-language-folders).",
|
||||
)
|
||||
def translate_command(
|
||||
language_codes,
|
||||
root_dir,
|
||||
@@ -106,6 +123,8 @@ def translate_command(
|
||||
min_confidence,
|
||||
add_disclaimer,
|
||||
repo_url,
|
||||
migrate_language_folders,
|
||||
dry_run,
|
||||
):
|
||||
"""
|
||||
CLI for translating project files.
|
||||
@@ -193,6 +212,8 @@ def translate_command(
|
||||
if save_logs and log_file_path is not None:
|
||||
click.echo(f"📄 Logs will be saved to: {log_file_path}")
|
||||
|
||||
# (Preview moved after language normalization)
|
||||
|
||||
# Now run the LLM health check; raises on failure
|
||||
LLMConfig.validate_connectivity()
|
||||
logger.info("LLM health check passed.")
|
||||
@@ -205,7 +226,7 @@ def translate_command(
|
||||
logger.info("Vision health check passed.")
|
||||
click.echo("✅ Vision health check passed.")
|
||||
|
||||
# Show warning if 'all' is selected
|
||||
# Normalize language codes and handle 'all'
|
||||
all_languages_selected = language_codes == "all"
|
||||
if all_languages_selected:
|
||||
click.echo(
|
||||
@@ -228,31 +249,76 @@ def translate_command(
|
||||
click.echo("Proceeding with translation for all languages...")
|
||||
else:
|
||||
click.echo("Auto-confirming translation for all languages...")
|
||||
# Use canonical list from config and normalize
|
||||
lang_list = Config.get_language_codes()
|
||||
if not lang_list:
|
||||
raise click.ClickException(
|
||||
"No valid language codes found in font mappings"
|
||||
)
|
||||
language_codes = " ".join(normalize_language_codes(lang_list))
|
||||
else:
|
||||
# Normalize explicit input codes to canonical form
|
||||
lang_list = normalize_language_codes(
|
||||
[code.strip() for code in language_codes.split()]
|
||||
)
|
||||
if not lang_list:
|
||||
raise click.ClickException("No valid language codes provided")
|
||||
language_codes = " ".join(lang_list)
|
||||
|
||||
try:
|
||||
with importlib.resources.path(
|
||||
"co_op_translator.fonts", "font_language_mappings.yml"
|
||||
) as mappings_path:
|
||||
with open(mappings_path, "r", encoding="utf-8") as file:
|
||||
font_mappings = yaml.safe_load(file)
|
||||
if not font_mappings:
|
||||
raise click.ClickException("Empty font mappings file")
|
||||
language_codes = " ".join(
|
||||
[
|
||||
lang_code
|
||||
for lang_code in font_mappings
|
||||
if isinstance(font_mappings[lang_code], dict)
|
||||
]
|
||||
)
|
||||
if not language_codes:
|
||||
raise click.ClickException(
|
||||
"No valid language codes found in font mappings"
|
||||
# Detect and migrate alias-based language folders for the SELECTED languages
|
||||
# This runs before any translation work to avoid redundant re-translation.
|
||||
try:
|
||||
migrator = LanguageFolderMigrator(root_path)
|
||||
alias_entries = migrator.detect_alias_folders()
|
||||
if alias_entries:
|
||||
# Filter only entries relevant to selected canonical languages
|
||||
relevant = [e for e in alias_entries if e.canonical in lang_list]
|
||||
if relevant:
|
||||
click.echo("\nMigration plan (selected languages):")
|
||||
click.echo(LanguageFolderMigrator.format_plan(relevant))
|
||||
if dry_run:
|
||||
click.echo("Dry-run: no changes will be made.")
|
||||
else:
|
||||
do_migrate = migrate_language_folders or yes
|
||||
if not do_migrate:
|
||||
# Ask for confirmation when not explicitly requested and not auto-confirmed
|
||||
confirm = click.prompt(
|
||||
"Migrate alias folders now? Type 'yes' to continue",
|
||||
type=str,
|
||||
default="no",
|
||||
)
|
||||
logging.debug(
|
||||
f"Loaded language codes from font mapping: {language_codes}"
|
||||
)
|
||||
except (FileNotFoundError, yaml.YAMLError) as e:
|
||||
raise click.ClickException(f"Failed to load font mappings: {str(e)}")
|
||||
do_migrate = confirm.strip().lower() == "yes"
|
||||
|
||||
if do_migrate:
|
||||
renamed, msgs = migrator.execute(relevant, dry_run=False)
|
||||
click.echo(f"Auto-migrate: renamed {renamed} folder(s).")
|
||||
for m in msgs:
|
||||
click.echo(f"- {m}")
|
||||
else:
|
||||
click.echo("Proceeding without migration.")
|
||||
else:
|
||||
if migrate_language_folders and dry_run:
|
||||
click.echo("No non-standard language folders detected.")
|
||||
except Exception as e:
|
||||
logger.warning(f"Language folder migration step skipped: {e}")
|
||||
|
||||
# Ensure per-language metadata files store canonical language_code values
|
||||
try:
|
||||
for lang in lang_list:
|
||||
# translations/<lang>/.co-op-translator.json
|
||||
normalize_language_codes_in_lang_metadata(
|
||||
root_path / "translations" / lang, lang
|
||||
)
|
||||
# translated_images/<lang>/.co-op-translator.json
|
||||
normalize_language_codes_in_lang_metadata(
|
||||
root_path / "translated_images" / lang, lang
|
||||
)
|
||||
# translated_images_fast/<lang>/.co-op-translator.json (best-effort)
|
||||
normalize_language_codes_in_lang_metadata(
|
||||
root_path / "translated_images_fast" / lang, lang
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"Metadata normalization skipped: {e}")
|
||||
|
||||
# Show deprecation warning when fast image mode is enabled
|
||||
if fast and "images" in translation_types:
|
||||
|
||||
@@ -6,6 +6,7 @@ import importlib.resources
|
||||
import yaml
|
||||
from co_op_translator.config.llm_config.config import LLMConfig
|
||||
from co_op_translator.config.vision_config.config import VisionConfig
|
||||
from co_op_translator.utils.common.lang_utils import normalize_language_code
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -46,7 +47,8 @@ class Config:
|
||||
@staticmethod
|
||||
def get_language_codes() -> list[str]:
|
||||
"""
|
||||
Return the full list of supported language codes from the packaged font mappings.
|
||||
Return the full list of supported language codes from the packaged font mappings,
|
||||
normalized to canonical BCP 47 (e.g., zh-TW, zh-CN, pt-BR, pt-PT).
|
||||
|
||||
Falls back to an empty list on error.
|
||||
"""
|
||||
@@ -56,12 +58,31 @@ class Config:
|
||||
) as mappings_path:
|
||||
with open(mappings_path, "r", encoding="utf-8") as file:
|
||||
font_mappings = yaml.safe_load(file) or {}
|
||||
# Only include entries with a dict mapping (same rule as CLI)
|
||||
return [
|
||||
lang_code
|
||||
for lang_code in font_mappings
|
||||
if isinstance(font_mappings[lang_code], dict)
|
||||
]
|
||||
# Only include entries with a dict mapping and normalize
|
||||
seen: set[str] = set()
|
||||
ordered: list[str] = []
|
||||
for key, meta in font_mappings.items():
|
||||
if not isinstance(meta, dict):
|
||||
continue
|
||||
key_str = str(key)
|
||||
canon = normalize_language_code(key_str)
|
||||
# Optional developer warning when a non-canonical key is present
|
||||
try:
|
||||
if os.getenv("COOP_STRICT_CANON_KEYS") == "1" and (
|
||||
not canon or canon != key_str
|
||||
):
|
||||
logger.warning(
|
||||
"Non-canonical or invalid language code '%s' in font_language_mappings.yml (normalized to '%s')",
|
||||
key_str,
|
||||
canon,
|
||||
)
|
||||
except Exception:
|
||||
# Do not fail on logging issues
|
||||
pass
|
||||
if canon and canon not in seen:
|
||||
seen.add(canon)
|
||||
ordered.append(canon)
|
||||
return ordered
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to load language codes from font mappings: {e}")
|
||||
return []
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
import importlib.resources
|
||||
import yaml
|
||||
from co_op_translator.utils.common.lang_utils import (
|
||||
normalize_language_code,
|
||||
)
|
||||
|
||||
|
||||
class FontConfig:
|
||||
@@ -13,6 +16,23 @@ class FontConfig:
|
||||
with open(mappings_path, "r", encoding="utf-8") as file:
|
||||
self.font_mappings = yaml.safe_load(file)
|
||||
|
||||
def _resolve_mapping_key(self, language_code: str) -> str:
|
||||
"""
|
||||
Resolve provided language code to a canonical BCP 47 key present in font_language_mappings.yml.
|
||||
The YAML now uses canonical keys only. Alias inputs are normalized before lookup.
|
||||
"""
|
||||
if not language_code:
|
||||
raise ValueError("Empty language code is not supported.")
|
||||
|
||||
# Normalize input (accept alias like "tw", "cn")
|
||||
canonical = normalize_language_code(language_code)
|
||||
if canonical in self.font_mappings:
|
||||
return canonical
|
||||
|
||||
raise ValueError(
|
||||
f"Font for language code '{language_code}' is not supported or not found."
|
||||
)
|
||||
|
||||
def get_font_path(self, language_code):
|
||||
"""
|
||||
Retrieve the font path for a given language code.
|
||||
@@ -26,7 +46,8 @@ class FontConfig:
|
||||
Raises:
|
||||
ValueError: If the language code or font is not found in the mappings.
|
||||
"""
|
||||
font_name = self.font_mappings.get(language_code, {}).get("font")
|
||||
key = self._resolve_mapping_key(language_code)
|
||||
font_name = self.font_mappings.get(key, {}).get("font")
|
||||
|
||||
if not font_name:
|
||||
raise ValueError(
|
||||
@@ -49,10 +70,12 @@ class FontConfig:
|
||||
Raises:
|
||||
ValueError: If the language code is not found in the mappings.
|
||||
"""
|
||||
if language_code not in self.font_mappings:
|
||||
try:
|
||||
key = self._resolve_mapping_key(language_code)
|
||||
except ValueError:
|
||||
# Preserve historical error message for tests/backward-compat
|
||||
raise ValueError(f"Language code '{language_code}' is not supported.")
|
||||
|
||||
return self.font_mappings.get(language_code, {}).get("name", language_code)
|
||||
return self.font_mappings.get(key, {}).get("name", key)
|
||||
|
||||
def is_rtl(self, language_code):
|
||||
"""
|
||||
@@ -67,8 +90,10 @@ class FontConfig:
|
||||
Raises:
|
||||
ValueError: If the language code is not found in the mappings.
|
||||
"""
|
||||
if language_code not in self.font_mappings:
|
||||
try:
|
||||
key = self._resolve_mapping_key(language_code)
|
||||
except ValueError:
|
||||
# Preserve historical error message for tests/backward-compat
|
||||
raise ValueError(f"Language code '{language_code}' is not supported.")
|
||||
|
||||
# Return RTL info if available, default to False
|
||||
return self.font_mappings.get(language_code, {}).get("rtl", False)
|
||||
return self.font_mappings.get(key, {}).get("rtl", False)
|
||||
|
||||
@@ -17,6 +17,7 @@ from co_op_translator.config.constants import (
|
||||
SUPPORTED_NOTEBOOK_EXTENSIONS,
|
||||
SUPPORTED_IMAGE_EXTENSIONS,
|
||||
)
|
||||
from co_op_translator.utils.common.lang_utils import normalize_language_code
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -436,16 +437,19 @@ class DirectoryManager:
|
||||
rel_parts = ()
|
||||
|
||||
lang_code = None
|
||||
if len(rel_parts) >= 2 and rel_parts[0] in self.language_codes:
|
||||
lang_code = rel_parts[0]
|
||||
# New format: base.hash.ext
|
||||
path_hash_segment = parts[-2]
|
||||
base_name = ".".join(parts[:-2])
|
||||
# Accept alias language folder names by normalizing to canonical
|
||||
if len(rel_parts) >= 2:
|
||||
parent_lang = rel_parts[0]
|
||||
normalized_parent = normalize_language_code(parent_lang)
|
||||
if normalized_parent in self.language_codes:
|
||||
lang_code = normalized_parent
|
||||
path_hash_segment = parts[-2]
|
||||
base_name = ".".join(parts[:-2])
|
||||
else:
|
||||
# Legacy format: base.hash.lang.ext
|
||||
if len(parts) < 4:
|
||||
continue
|
||||
lang_code = parts[-2]
|
||||
lang_code = normalize_language_code(parts[-2])
|
||||
path_hash_segment = parts[-3]
|
||||
base_name = ".".join(parts[:-3])
|
||||
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import List, Tuple
|
||||
|
||||
from co_op_translator.utils.common.lang_utils import ALIAS_TO_BCP47
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class MigrationEntry:
|
||||
base_dir: Path
|
||||
source_dir: Path
|
||||
dest_dir: Path
|
||||
alias: str
|
||||
canonical: str
|
||||
conflict: bool = False
|
||||
|
||||
|
||||
class LanguageFolderMigrator:
|
||||
def __init__(
|
||||
self,
|
||||
root_dir: Path,
|
||||
translations_dir: Path | None = None,
|
||||
image_dir: Path | None = None,
|
||||
):
|
||||
self.root_dir = Path(root_dir)
|
||||
self.translations_dir = (
|
||||
(self.root_dir / "translations")
|
||||
if translations_dir is None
|
||||
else Path(translations_dir)
|
||||
)
|
||||
self.image_dir = (
|
||||
(self.root_dir / "translated_images")
|
||||
if image_dir is None
|
||||
else Path(image_dir)
|
||||
)
|
||||
self.image_fast_dir = self.root_dir / "translated_images_fast"
|
||||
self._aliases = set(ALIAS_TO_BCP47.keys())
|
||||
|
||||
def detect_alias_folders(self) -> List[MigrationEntry]:
|
||||
entries: List[MigrationEntry] = []
|
||||
for base in [self.translations_dir, self.image_dir, self.image_fast_dir]:
|
||||
if not base.exists() or not base.is_dir():
|
||||
continue
|
||||
try:
|
||||
for child in base.iterdir():
|
||||
if not child.is_dir():
|
||||
continue
|
||||
alias = child.name
|
||||
if alias in self._aliases:
|
||||
canonical = ALIAS_TO_BCP47[alias]
|
||||
dest = base / canonical
|
||||
conflict = dest.exists() and any(dest.iterdir())
|
||||
entries.append(
|
||||
MigrationEntry(
|
||||
base_dir=base,
|
||||
source_dir=child,
|
||||
dest_dir=dest,
|
||||
alias=alias,
|
||||
canonical=canonical,
|
||||
conflict=conflict,
|
||||
)
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed scanning {base}: {e}")
|
||||
return entries
|
||||
|
||||
@staticmethod
|
||||
def format_plan(entries: List[MigrationEntry]) -> str:
|
||||
if not entries:
|
||||
return "No non-standard language folders detected."
|
||||
lines: List[str] = ["Detected non-standard language folders:"]
|
||||
for e in entries:
|
||||
status = "(conflict)" if e.conflict else ""
|
||||
lines.append(
|
||||
f"- {e.source_dir.relative_to(e.base_dir)} -> {e.dest_dir.relative_to(e.base_dir)} {status}"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
|
||||
def _fs_rename(self, src: Path, dst: Path) -> Tuple[bool, str]:
|
||||
try:
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
src.rename(dst)
|
||||
return True, ""
|
||||
except Exception as e:
|
||||
return False, str(e)
|
||||
|
||||
def _rewrite_lang_metadata(self, lang_dir: Path, canonical_code: str) -> None:
|
||||
"""Ensure per-language metadata under lang_dir stores canonical language_code.
|
||||
|
||||
Delegates to normalize_language_codes_in_lang_metadata to avoid duplication.
|
||||
"""
|
||||
try:
|
||||
from co_op_translator.utils.common.metadata_utils import (
|
||||
normalize_language_codes_in_lang_metadata,
|
||||
)
|
||||
|
||||
normalize_language_codes_in_lang_metadata(lang_dir, canonical_code)
|
||||
except Exception:
|
||||
# Non-fatal if metadata normalization fails
|
||||
pass
|
||||
|
||||
def execute(
|
||||
self, entries: List[MigrationEntry], use_git: bool | None = None, dry_run: bool = False
|
||||
) -> Tuple[int, List[str]]:
|
||||
"""Execute migration entries. Skips conflicting destinations.
|
||||
|
||||
Returns number of successful renames and a list of messages for errors or conflicts.
|
||||
|
||||
Note: `use_git` is currently ignored and preserved only for backward compatibility.
|
||||
"""
|
||||
if not entries:
|
||||
return 0, []
|
||||
|
||||
msgs: List[str] = []
|
||||
successes = 0
|
||||
for e in entries:
|
||||
if e.conflict:
|
||||
msgs.append(
|
||||
f"Conflict: '{e.dest_dir}' already exists. Skipping auto-merge for '{e.source_dir}'."
|
||||
)
|
||||
continue
|
||||
if dry_run:
|
||||
# No changes, preview only
|
||||
continue
|
||||
ok, err = self._fs_rename(e.source_dir, e.dest_dir)
|
||||
if ok:
|
||||
successes += 1
|
||||
# After successful rename, rewrite metadata to canonical code
|
||||
try:
|
||||
self._rewrite_lang_metadata(e.dest_dir, e.canonical)
|
||||
except Exception:
|
||||
# Non-fatal if metadata rewrite fails
|
||||
pass
|
||||
else:
|
||||
msgs.append(
|
||||
f"Failed to rename '{e.source_dir}' -> '{e.dest_dir}': {err}"
|
||||
)
|
||||
return successes, msgs
|
||||
@@ -14,6 +14,11 @@ from co_op_translator.config.constants import (
|
||||
SUPPORTED_IMAGE_EXTENSIONS,
|
||||
SUPPORTED_NOTEBOOK_EXTENSIONS,
|
||||
)
|
||||
from co_op_translator.utils.common.lang_utils import (
|
||||
normalize_language_codes,
|
||||
get_supported_language_codes,
|
||||
ALIAS_TO_BCP47,
|
||||
)
|
||||
|
||||
from .directory_manager import DirectoryManager
|
||||
from .translation_manager import TranslationManager
|
||||
@@ -46,7 +51,8 @@ class ProjectTranslator:
|
||||
root_dir: Root directory of the project to translate
|
||||
translation_types: List of file types to translate (e.g., ["markdown", "images", "notebook"])
|
||||
"""
|
||||
self.language_codes = language_codes.split()
|
||||
# Normalize to canonical BCP 47 (accept alias input like tw/cn/br)
|
||||
self.language_codes = normalize_language_codes(language_codes.split())
|
||||
self.root_dir = Path(root_dir).resolve()
|
||||
# Resolve translations_dir relative to root_dir when a relative path is provided.
|
||||
if translations_dir is not None:
|
||||
@@ -83,6 +89,21 @@ class ProjectTranslator:
|
||||
except ValueError:
|
||||
# Directory is outside root_dir; no need to exclude from root scans
|
||||
continue
|
||||
# Dynamically add root-level language code folders (canonical or alias) to exclusions
|
||||
try:
|
||||
present_subdirs = [p for p in self.root_dir.iterdir() if p.is_dir()]
|
||||
except Exception:
|
||||
present_subdirs = []
|
||||
supported = set(get_supported_language_codes())
|
||||
aliases = set(ALIAS_TO_BCP47.keys())
|
||||
for sub in present_subdirs:
|
||||
name = sub.name
|
||||
if name in supported or name in aliases:
|
||||
# Only add absolute path strings to avoid substring false-positives in DirectoryManager
|
||||
try:
|
||||
excluded_dirs.add(str(sub.resolve()))
|
||||
except Exception:
|
||||
excluded_dirs.add(str(sub))
|
||||
self.excluded_dirs = list(excluded_dirs)
|
||||
|
||||
# Initialize text translator
|
||||
|
||||
@@ -35,6 +35,9 @@ from co_op_translator.utils.common.task_utils import worker
|
||||
from co_op_translator.utils.llm.markdown_utils import compare_line_breaks
|
||||
from co_op_translator.utils.common.metadata_utils import is_notebook_up_to_date
|
||||
from co_op_translator.config.base_config import Config
|
||||
from co_op_translator.utils.common.file_utils import (
|
||||
canonicalize_image_links_in_translations,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -662,6 +665,20 @@ class TranslationManager:
|
||||
migrated_md,
|
||||
migrated_nb,
|
||||
)
|
||||
|
||||
# As a safety net, canonicalize any remaining alias-based language dir segments in links
|
||||
try:
|
||||
md_fix, nb_fix = canonicalize_image_links_in_translations(
|
||||
self.translations_dir, self.image_dir
|
||||
)
|
||||
if md_fix or nb_fix:
|
||||
logger.info(
|
||||
"Canonicalized image links in %d markdown and %d notebooks",
|
||||
md_fix,
|
||||
nb_fix,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"Image link canonicalization skipped: {e}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Image filename/link migration skipped: {e}")
|
||||
|
||||
|
||||
@@ -25,16 +25,16 @@ ur:
|
||||
name: "Urdu"
|
||||
font: "NotoSansArabic-Medium.ttf"
|
||||
rtl: true
|
||||
zh:
|
||||
zh-CN:
|
||||
name: "Chinese (Simplified)"
|
||||
font: "NotoSansCJK-Medium.ttc"
|
||||
mo:
|
||||
zh-MO:
|
||||
name: "Chinese (Traditional, Macau)"
|
||||
font: "NotoSansCJK-Medium.ttc"
|
||||
hk:
|
||||
zh-HK:
|
||||
name: "Chinese (Traditional, Hong Kong)"
|
||||
font: "NotoSansCJK-Medium.ttc"
|
||||
tw:
|
||||
zh-TW:
|
||||
name: "Chinese (Traditional, Taiwan)"
|
||||
font: "NotoSansCJK-Medium.ttc"
|
||||
ja:
|
||||
@@ -58,11 +58,11 @@ ne:
|
||||
pa:
|
||||
name: "Punjabi (Gurmukhi)"
|
||||
font: "NotoSansGurmukhi-Medium.ttf"
|
||||
pt:
|
||||
pt-PT:
|
||||
name: "Portuguese (Portugal)"
|
||||
font: "NotoSans-Medium.ttf"
|
||||
|
||||
br:
|
||||
pt-BR:
|
||||
name: "Portuguese (Brazil)"
|
||||
font: "NotoSans-Medium.ttf"
|
||||
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
<!-- CO-OP TRANSLATOR LANGUAGES TABLE START -->
|
||||
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh/README.md) | [Chinese (Traditional, Hong Kong)](./translations/hk/README.md) | [Chinese (Traditional, Macau)](./translations/mo/README.md) | [Chinese (Traditional, Taiwan)](./translations/tw/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Kannada](./translations/kn/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Malayalam](./translations/ml/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Nigerian Pidgin](./translations/pcm/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/br/README.md) | [Portuguese (Portugal)](./translations/pt/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Telugu](./translations/te/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
|
||||
[Arabic](./translations/ar/README.md) | [Bengali](./translations/bn/README.md) | [Bulgarian](./translations/bg/README.md) | [Burmese (Myanmar)](./translations/my/README.md) | [Chinese (Simplified)](./translations/zh-CN/README.md) | [Chinese (Traditional, Hong Kong)](./translations/zh-HK/README.md) | [Chinese (Traditional, Macau)](./translations/zh-MO/README.md) | [Chinese (Traditional, Taiwan)](./translations/zh-TW/README.md) | [Croatian](./translations/hr/README.md) | [Czech](./translations/cs/README.md) | [Danish](./translations/da/README.md) | [Dutch](./translations/nl/README.md) | [Estonian](./translations/et/README.md) | [Finnish](./translations/fi/README.md) | [French](./translations/fr/README.md) | [German](./translations/de/README.md) | [Greek](./translations/el/README.md) | [Hebrew](./translations/he/README.md) | [Hindi](./translations/hi/README.md) | [Hungarian](./translations/hu/README.md) | [Indonesian](./translations/id/README.md) | [Italian](./translations/it/README.md) | [Japanese](./translations/ja/README.md) | [Kannada](./translations/kn/README.md) | [Korean](./translations/ko/README.md) | [Lithuanian](./translations/lt/README.md) | [Malay](./translations/ms/README.md) | [Malayalam](./translations/ml/README.md) | [Marathi](./translations/mr/README.md) | [Nepali](./translations/ne/README.md) | [Nigerian Pidgin](./translations/pcm/README.md) | [Norwegian](./translations/no/README.md) | [Persian (Farsi)](./translations/fa/README.md) | [Polish](./translations/pl/README.md) | [Portuguese (Brazil)](./translations/pt-BR/README.md) | [Portuguese (Portugal)](./translations/pt-PT/README.md) | [Punjabi (Gurmukhi)](./translations/pa/README.md) | [Romanian](./translations/ro/README.md) | [Russian](./translations/ru/README.md) | [Serbian (Cyrillic)](./translations/sr/README.md) | [Slovak](./translations/sk/README.md) | [Slovenian](./translations/sl/README.md) | [Spanish](./translations/es/README.md) | [Swahili](./translations/sw/README.md) | [Swedish](./translations/sv/README.md) | [Tagalog (Filipino)](./translations/tl/README.md) | [Tamil](./translations/ta/README.md) | [Telugu](./translations/te/README.md) | [Thai](./translations/th/README.md) | [Turkish](./translations/tr/README.md) | [Ukrainian](./translations/uk/README.md) | [Urdu](./translations/ur/README.md) | [Vietnamese](./translations/vi/README.md)
|
||||
<!-- CO-OP TRANSLATOR LANGUAGES TABLE END -->
|
||||
@@ -423,15 +423,49 @@ def filter_files(directory: str | Path, excluded_dirs, extension: str = None) ->
|
||||
directory = Path(directory)
|
||||
files = []
|
||||
|
||||
# Normalize excluded directories into two buckets: absolute paths and names
|
||||
abs_excluded: list[Path] = []
|
||||
name_excluded: set[str] = set()
|
||||
for item in excluded_dirs:
|
||||
try:
|
||||
p = Path(item)
|
||||
if p.is_absolute():
|
||||
abs_excluded.append(p.resolve())
|
||||
else:
|
||||
name_excluded.add(str(item))
|
||||
except Exception:
|
||||
name_excluded.add(str(item))
|
||||
|
||||
# Recursively traverse the directory
|
||||
for path in directory.rglob("*"):
|
||||
# Check if the path is a file, matches extension if specified, and doesn't contain excluded dirs
|
||||
if (
|
||||
path.is_file()
|
||||
and (extension is None or path.suffix.lower() == extension.lower())
|
||||
and not any(excluded_dir in path.parts for excluded_dir in excluded_dirs)
|
||||
):
|
||||
files.append(path)
|
||||
if not path.is_file():
|
||||
continue
|
||||
if extension is not None and path.suffix.lower() != extension.lower():
|
||||
continue
|
||||
|
||||
# Exclude by absolute path ancestry
|
||||
excluded_by_abs = False
|
||||
try:
|
||||
resolved = path.resolve()
|
||||
for abs_dir in abs_excluded:
|
||||
try:
|
||||
resolved.relative_to(abs_dir)
|
||||
excluded_by_abs = True
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
except Exception:
|
||||
# If resolve() fails, fall back to name-based exclusion only
|
||||
excluded_by_abs = False
|
||||
|
||||
if excluded_by_abs:
|
||||
continue
|
||||
|
||||
# Name-based exclusion (segment match)
|
||||
if any(ex in path.parts for ex in name_excluded):
|
||||
continue
|
||||
|
||||
files.append(path)
|
||||
|
||||
return files
|
||||
|
||||
@@ -462,6 +496,8 @@ def migrate_translated_image_filenames(
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
from co_op_translator.utils.common.lang_utils import normalize_language_code
|
||||
|
||||
for image_file in image_files:
|
||||
if not image_file.is_file():
|
||||
continue
|
||||
@@ -486,18 +522,26 @@ def migrate_translated_image_filenames(
|
||||
extension = parts[-1]
|
||||
|
||||
# Detect language either from directory or from legacy filename
|
||||
under_lang_dir = len(rel_parts) >= 2 and rel_parts[0] in language_codes
|
||||
lang_code: str | None = rel_parts[0] if under_lang_dir else None
|
||||
under_lang_dir = False
|
||||
lang_code: str | None = None
|
||||
if len(rel_parts) >= 2:
|
||||
parent_lang = rel_parts[0]
|
||||
normalized_parent = normalize_language_code(parent_lang)
|
||||
if normalized_parent in language_codes:
|
||||
under_lang_dir = True
|
||||
lang_code = normalized_parent
|
||||
|
||||
# Legacy filename pattern includes trailing language segment
|
||||
has_legacy_lang_in_name = len(parts) >= 4 and parts[-2] in language_codes
|
||||
has_legacy_lang_in_name = (
|
||||
len(parts) >= 4 and normalize_language_code(parts[-2]) in language_codes
|
||||
)
|
||||
|
||||
if not under_lang_dir and not has_legacy_lang_in_name:
|
||||
# Cannot determine language; skip conservatively
|
||||
continue
|
||||
|
||||
if has_legacy_lang_in_name and lang_code is None:
|
||||
lang_code = parts[-2]
|
||||
lang_code = normalize_language_code(parts[-2])
|
||||
|
||||
if lang_code not in language_codes:
|
||||
# Skip unsupported languages
|
||||
@@ -801,3 +845,88 @@ def delete_translated_markdown_files_by_language_code(
|
||||
# Remove the entire directory and its contents
|
||||
shutil.rmtree(language_dir)
|
||||
logger.info(f"Deleted the directory and all files for language: {language_code}")
|
||||
|
||||
|
||||
def canonicalize_image_links_in_translations(
|
||||
translations_dir: Path, image_dir: Path
|
||||
) -> tuple[int, int]:
|
||||
"""
|
||||
Canonicalize image links in translated markdown and notebooks by rewriting
|
||||
alias-based language directory segments to canonical BCP 47.
|
||||
|
||||
Examples:
|
||||
translated_images/tw/... -> translated_images/zh-TW/...
|
||||
translated_images/cn/... -> translated_images/zh-CN/...
|
||||
<base_dir>/br/... -> <base_dir>/pt-BR/...
|
||||
|
||||
The function scans under translations_dir and updates files in-place.
|
||||
|
||||
Returns:
|
||||
(md_files_updated, nb_files_updated)
|
||||
"""
|
||||
from co_op_translator.utils.common.lang_utils import ALIAS_TO_BCP47
|
||||
from co_op_translator.config.constants import (
|
||||
SUPPORTED_MARKDOWN_EXTENSIONS,
|
||||
SUPPORTED_NOTEBOOK_EXTENSIONS,
|
||||
)
|
||||
|
||||
translations_dir = Path(translations_dir)
|
||||
image_dir = Path(image_dir)
|
||||
base_dir_name = image_dir.name
|
||||
base_dirs = [base_dir_name, "translated_images", "translated_images_fast"]
|
||||
|
||||
def _canonicalize_text(text: str) -> str:
|
||||
updated = text
|
||||
for bdir in base_dirs:
|
||||
for alias, canonical in ALIAS_TO_BCP47.items():
|
||||
updated = updated.replace(f"{bdir}/{alias}/", f"{bdir}/{canonical}/")
|
||||
# Also replace Windows-style separators just in case
|
||||
updated = updated.replace(
|
||||
f"{bdir}\\{alias}\\", f"{bdir}\\{canonical}\\"
|
||||
)
|
||||
return updated
|
||||
|
||||
md_updated = 0
|
||||
nb_updated = 0
|
||||
|
||||
# Markdown files
|
||||
try:
|
||||
md_files: list[Path] = []
|
||||
for ext in SUPPORTED_MARKDOWN_EXTENSIONS:
|
||||
md_files.extend(translations_dir.rglob(f"*{ext}"))
|
||||
for md in md_files:
|
||||
try:
|
||||
original = md.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
continue
|
||||
updated = _canonicalize_text(original)
|
||||
if updated != original:
|
||||
try:
|
||||
md.write_text(updated, encoding="utf-8")
|
||||
md_updated += 1
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Notebooks (JSON)
|
||||
try:
|
||||
nb_files: list[Path] = []
|
||||
for ext in SUPPORTED_NOTEBOOK_EXTENSIONS:
|
||||
nb_files.extend(translations_dir.rglob(f"*{ext}"))
|
||||
for nb in nb_files:
|
||||
try:
|
||||
content = nb.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
continue
|
||||
updated = _canonicalize_text(content)
|
||||
if updated != content:
|
||||
try:
|
||||
nb.write_text(updated, encoding="utf-8")
|
||||
nb_updated += 1
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return md_updated, nb_updated
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.resources
|
||||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Tuple
|
||||
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Explicit alias map only (no heuristics). All keys are lower-case.
|
||||
ALIAS_TO_BCP47: dict[str, str] = {
|
||||
# Chinese regions
|
||||
"cn": "zh-CN",
|
||||
"tw": "zh-TW",
|
||||
"hk": "zh-HK",
|
||||
"mo": "zh-MO",
|
||||
# Portuguese regions
|
||||
"br": "pt-BR",
|
||||
"pt": "pt-PT", # treat bare 'pt' as Portugal when normalizing
|
||||
# Common country-code aliases
|
||||
"jp": "ja",
|
||||
"kr": "ko",
|
||||
# Generic zh alias maps to Mainland China for our purposes
|
||||
"zh": "zh-CN",
|
||||
}
|
||||
|
||||
# BCP 47 basic pattern: language[-script][-region][-variants...]
|
||||
# We only standardize language + region casing here, not validating scripts/variants.
|
||||
_BCP47_SPLIT = re.compile(r"[-_]")
|
||||
|
||||
|
||||
def canonical_case(code: str) -> str:
|
||||
"""Apply canonical casing: language lower-case, region upper-case.
|
||||
|
||||
Examples:
|
||||
- "pt-br" -> "pt-BR"
|
||||
- "ZH-tw" -> "zh-TW"
|
||||
- "ja" -> "ja"
|
||||
"""
|
||||
parts = _BCP47_SPLIT.split(code.strip())
|
||||
if not parts:
|
||||
return code
|
||||
lang = parts[0].lower()
|
||||
rest: List[str] = []
|
||||
for i, p in enumerate(parts[1:], start=1):
|
||||
if len(p) == 2: # region
|
||||
rest.append(p.upper())
|
||||
else:
|
||||
# leave script/variants as-is but prefer title for script (4 letters)
|
||||
if len(p) == 4:
|
||||
rest.append(p.title())
|
||||
else:
|
||||
rest.append(p)
|
||||
return "-".join([lang, *rest]) if rest else lang
|
||||
|
||||
|
||||
def normalize_language_code(code: str) -> str:
|
||||
"""Normalize an input code to our canonical BCP 47 form using explicit aliases.
|
||||
|
||||
- Applies explicit alias mapping first (exact match on lower-cased input)
|
||||
- Then applies canonical casing rules.
|
||||
"""
|
||||
raw = code.strip()
|
||||
if not raw:
|
||||
return raw
|
||||
key = raw.lower()
|
||||
if key in ALIAS_TO_BCP47:
|
||||
return ALIAS_TO_BCP47[key]
|
||||
return canonical_case(raw)
|
||||
|
||||
|
||||
def normalize_language_codes(codes: Iterable[str]) -> List[str]:
|
||||
seen: set[str] = set()
|
||||
normalized: List[str] = []
|
||||
for c in codes:
|
||||
canon = normalize_language_code(c)
|
||||
if canon not in seen and canon:
|
||||
seen.add(canon)
|
||||
normalized.append(canon)
|
||||
return normalized
|
||||
|
||||
|
||||
def get_supported_language_codes() -> List[str]:
|
||||
"""Return canonical supported codes (keys) from font_language_mappings.yml.
|
||||
|
||||
Note: This list reflects our packaging and is used when "all" is selected.
|
||||
"""
|
||||
try:
|
||||
with importlib.resources.path(
|
||||
"co_op_translator.fonts", "font_language_mappings.yml"
|
||||
) as mappings_path:
|
||||
with open(mappings_path, "r", encoding="utf-8") as file:
|
||||
font_mappings = yaml.safe_load(file) or {}
|
||||
return [
|
||||
lang_code
|
||||
for lang_code, meta in font_mappings.items()
|
||||
if isinstance(meta, dict)
|
||||
]
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to load font mappings: {e}")
|
||||
return []
|
||||
|
||||
|
||||
def is_supported_language(code: str) -> bool:
|
||||
canon = normalize_language_code(code)
|
||||
return canon in set(get_supported_language_codes())
|
||||
@@ -671,3 +671,35 @@ def remove_text_metadata_for_source(lang_dir: Path, source_file: str | Path) ->
|
||||
pass
|
||||
if changed:
|
||||
_save_lang_metadata(lang_dir, all_metadata)
|
||||
|
||||
|
||||
def normalize_language_codes_in_lang_metadata(
|
||||
lang_dir: Path, canonical_code: str
|
||||
) -> int:
|
||||
"""Normalize 'language_code' fields in .co-op-translator.json under a language folder.
|
||||
|
||||
Args:
|
||||
lang_dir: Path to translations/<lang> or translated_images/<lang> directory
|
||||
canonical_code: Expected canonical BCP47 language code (e.g., 'pt-BR')
|
||||
|
||||
Returns:
|
||||
Number of entries updated.
|
||||
"""
|
||||
lang_dir = Path(lang_dir)
|
||||
metadata_path = _get_metadata_file_path(lang_dir)
|
||||
if not metadata_path.exists():
|
||||
return 0
|
||||
try:
|
||||
data = _load_lang_metadata(lang_dir)
|
||||
if not isinstance(data, dict) or not data:
|
||||
return 0
|
||||
changed = 0
|
||||
for k, v in list(data.items()):
|
||||
if isinstance(v, dict) and v.get("language_code") != canonical_code:
|
||||
v["language_code"] = canonical_code
|
||||
changed += 1
|
||||
if changed:
|
||||
_save_lang_metadata(lang_dir, data)
|
||||
return changed
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch, mock_open
|
||||
|
||||
from co_op_translator.config.font_config import FontConfig
|
||||
|
||||
sample_yaml = """
|
||||
zh-TW:
|
||||
name: Chinese (Traditional, Taiwan)
|
||||
font: "NotoSansCJK-Medium.ttc"
|
||||
pt-PT:
|
||||
name: Portuguese (Portugal)
|
||||
font: "NotoSans-Medium.ttf"
|
||||
pt-BR:
|
||||
name: Portuguese (Brazil)
|
||||
font: "NotoSans-Medium.ttf"
|
||||
"""
|
||||
|
||||
|
||||
def test_font_config_resolves_canonical_to_alias_keys():
|
||||
# Mock YAML with alias keys only
|
||||
with (
|
||||
patch("importlib.resources.path") as mock_path_yaml,
|
||||
patch("builtins.open", mock_open(read_data=sample_yaml)),
|
||||
):
|
||||
mock_path_yaml.return_value = Path("fake/fonts/font_language_mappings.yml")
|
||||
fc = FontConfig()
|
||||
|
||||
# get_font_path should resolve alias input 'tw' to canonical 'zh-TW'
|
||||
with patch(
|
||||
"importlib.resources.path",
|
||||
return_value=Path("fake_fonts/NotoSansCJK-Medium.ttc"),
|
||||
) as mock_path_font:
|
||||
path = fc.get_font_path("tw")
|
||||
assert Path(path).name == "NotoSansCJK-Medium.ttc"
|
||||
mock_path_font.assert_called_once_with(
|
||||
"co_op_translator.fonts", "NotoSansCJK-Medium.ttc"
|
||||
)
|
||||
|
||||
# get_language_name should resolve alias input 'br' to canonical 'pt-BR'
|
||||
name = fc.get_language_name("br")
|
||||
assert name == "Portuguese (Brazil)"
|
||||
|
||||
# is_rtl defaults to False if not set
|
||||
assert fc.is_rtl("zh-TW") is False
|
||||
|
||||
|
||||
def test_font_config_invalid_language_errors():
|
||||
with (
|
||||
patch("importlib.resources.path") as mock_path_yaml,
|
||||
patch("builtins.open", mock_open(read_data=sample_yaml)),
|
||||
):
|
||||
mock_path_yaml.return_value = Path("fake/fonts/font_language_mappings.yml")
|
||||
fc = FontConfig()
|
||||
|
||||
with pytest.raises(ValueError) as excinfo:
|
||||
fc.get_language_name("xx")
|
||||
assert "Language code 'xx' is not supported." in str(excinfo.value)
|
||||
|
||||
with pytest.raises(ValueError) as excinfo2:
|
||||
fc.is_rtl("xx")
|
||||
assert "Language code 'xx' is not supported." in str(excinfo2.value)
|
||||
@@ -118,7 +118,7 @@ class TestDirectoryManager:
|
||||
# Create translated image with hash under image_dir (default: root_dir/translated_images)
|
||||
translations_dir.mkdir(exist_ok=True)
|
||||
image_dir = root_dir / "translated_images"
|
||||
ko_img_dir = image_dir / "img"
|
||||
ko_img_dir = image_dir / "ko" / "img"
|
||||
ko_img_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Valid translated image (with correct hash)
|
||||
@@ -126,12 +126,13 @@ class TestDirectoryManager:
|
||||
|
||||
original_name, ext = os.path.splitext(original_img.name)
|
||||
path_hash = get_unique_id(original_img, root_dir)
|
||||
valid_trans_name = f"{original_name}.{path_hash}.ko{ext}"
|
||||
# Canonical layout: translated_images/<lang>/<basename>.<hash>.<ext>
|
||||
valid_trans_name = f"{original_name}.{path_hash}{ext}"
|
||||
valid_trans_img = ko_img_dir / valid_trans_name
|
||||
valid_trans_img.write_bytes(b"translated content")
|
||||
|
||||
# Orphaned translated image (with incorrect hash)
|
||||
orphaned_trans_img = ko_img_dir / "test.invalid_hash.ko.png"
|
||||
orphaned_trans_img = ko_img_dir / "test.invalid_hash.png"
|
||||
orphaned_trans_img.write_bytes(b"orphaned content")
|
||||
|
||||
manager = DirectoryManager(
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
from pathlib import Path
|
||||
from co_op_translator.core.project.project_translator import ProjectTranslator
|
||||
from co_op_translator.utils.common.file_utils import filter_files
|
||||
|
||||
|
||||
def test_dynamic_exclusion_of_root_language_dirs(tmp_path: Path):
|
||||
root = tmp_path
|
||||
# Create root-level language dirs (canonical and alias) and a normal docs folder
|
||||
(root / "zh-TW").mkdir()
|
||||
(root / "cn").mkdir()
|
||||
(root / "docs").mkdir()
|
||||
|
||||
# Create files
|
||||
(root / "zh-TW" / "a.md").write_text("x", encoding="utf-8")
|
||||
(root / "cn" / "b.md").write_text("y", encoding="utf-8")
|
||||
(root / "docs" / "c.md").write_text("z", encoding="utf-8")
|
||||
|
||||
# Initialize translator (this will compute dynamic exclusions)
|
||||
pt = ProjectTranslator(
|
||||
language_codes="ko",
|
||||
root_dir=str(root),
|
||||
translation_types=["markdown"],
|
||||
)
|
||||
|
||||
files = filter_files(root, pt.excluded_dirs, extension=".md")
|
||||
paths = {str(p.relative_to(root)).replace("\\", "/") for p in files}
|
||||
|
||||
# Only docs/c.md should be present; language-dir files must be excluded
|
||||
assert "docs/c.md" in paths
|
||||
assert "zh-TW/a.md" not in paths
|
||||
assert "cn/b.md" not in paths
|
||||
@@ -0,0 +1,55 @@
|
||||
from pathlib import Path
|
||||
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
|
||||
|
||||
|
||||
def test_detect_alias_folders(tmp_path: Path):
|
||||
# Arrange a fake project with alias folders
|
||||
root = tmp_path
|
||||
translations = root / "translations"
|
||||
images = root / "translated_images"
|
||||
fast = root / "translated_images_fast"
|
||||
|
||||
(translations / "tw").mkdir(parents=True)
|
||||
(images / "cn").mkdir(parents=True)
|
||||
(fast / "br").mkdir(parents=True)
|
||||
|
||||
migrator = LanguageFolderMigrator(root_dir=root)
|
||||
|
||||
# Act
|
||||
entries = migrator.detect_alias_folders()
|
||||
|
||||
# Assert
|
||||
mapping = {(e.base_dir.name, e.alias, e.canonical) for e in entries}
|
||||
assert ("translations", "tw", "zh-TW") in mapping
|
||||
assert ("translated_images", "cn", "zh-CN") in mapping
|
||||
assert ("translated_images_fast", "br", "pt-BR") in mapping
|
||||
|
||||
|
||||
def test_execute_fs_rename(tmp_path: Path):
|
||||
# Arrange alias source and missing canonical destination
|
||||
root = tmp_path
|
||||
translations = root / "translations"
|
||||
(translations / "tw").mkdir(parents=True)
|
||||
(translations / "tw" / "foo").mkdir()
|
||||
|
||||
migrator = LanguageFolderMigrator(root_dir=root)
|
||||
entries = migrator.detect_alias_folders()
|
||||
|
||||
# Precondition
|
||||
assert (translations / "tw").exists()
|
||||
assert not (translations / "zh-TW").exists()
|
||||
|
||||
# Act: execute with dry_run first
|
||||
renamed, msgs = migrator.execute(entries, use_git=False, dry_run=True)
|
||||
# No changes in dry run
|
||||
assert renamed == 0
|
||||
assert (translations / "tw").exists()
|
||||
assert not (translations / "zh-TW").exists()
|
||||
|
||||
# Act: real rename (filesystem)
|
||||
renamed, msgs = migrator.execute(entries, use_git=False, dry_run=False)
|
||||
|
||||
# Assert rename happened
|
||||
assert renamed == 1
|
||||
assert not (translations / "tw").exists()
|
||||
assert (translations / "zh-TW").exists()
|
||||
@@ -0,0 +1,113 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from co_op_translator.utils.common.file_utils import (
|
||||
canonicalize_image_links_in_translations,
|
||||
)
|
||||
|
||||
|
||||
def write_text(p: Path, content: str) -> None:
|
||||
p.parent.mkdir(parents=True, exist_ok=True)
|
||||
p.write_text(content, encoding="utf-8")
|
||||
|
||||
|
||||
def test_markdown_rewrites_aliases_for_multiple_bases(tmp_path: Path):
|
||||
translations_dir = tmp_path / "translations"
|
||||
image_dir = tmp_path / "images_out" # custom base dir name
|
||||
|
||||
# Create a markdown file under translations with multiple alias patterns
|
||||
md = translations_dir / "ko" / "docs" / "sample.md"
|
||||
original = (
|
||||
"\n"
|
||||
"\n"
|
||||
f"\n"
|
||||
)
|
||||
write_text(md, original)
|
||||
|
||||
md_updated, nb_updated = canonicalize_image_links_in_translations(
|
||||
translations_dir=translations_dir, image_dir=image_dir
|
||||
)
|
||||
|
||||
# One markdown file updated, no notebooks
|
||||
assert md_updated == 1
|
||||
assert nb_updated == 0
|
||||
|
||||
updated = md.read_text(encoding="utf-8")
|
||||
assert "translated_images/zh-TW/foo.webp" in updated
|
||||
assert "translated_images_fast/zh-CN/bar.webp" in updated
|
||||
assert f"{image_dir.name}/pt-BR/baz.webp" in updated
|
||||
|
||||
|
||||
def test_notebook_rewrites_aliases(tmp_path: Path):
|
||||
translations_dir = tmp_path / "translations"
|
||||
image_dir = tmp_path / "translated_images" # common default base
|
||||
|
||||
nb = translations_dir / "ja" / "nb" / "sample.ipynb"
|
||||
nb_content = {
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"\n",
|
||||
"text before ",
|
||||
"\n",
|
||||
],
|
||||
}
|
||||
],
|
||||
"metadata": {},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5,
|
||||
}
|
||||
nb.parent.mkdir(parents=True, exist_ok=True)
|
||||
nb.write_text(json.dumps(nb_content), encoding="utf-8")
|
||||
|
||||
md_updated, nb_updated = canonicalize_image_links_in_translations(
|
||||
translations_dir=translations_dir, image_dir=image_dir
|
||||
)
|
||||
|
||||
assert md_updated == 0
|
||||
assert nb_updated == 1
|
||||
|
||||
updated = nb.read_text(encoding="utf-8")
|
||||
assert "translated_images/zh-CN/a.webp" in updated
|
||||
assert "translated_images_fast/pt-BR/b.webp" in updated
|
||||
|
||||
|
||||
def test_no_changes_returns_zero(tmp_path: Path):
|
||||
translations_dir = tmp_path / "translations"
|
||||
image_dir = tmp_path / "images_out"
|
||||
|
||||
md = translations_dir / "fr" / "doc.md"
|
||||
write_text(md, "No image paths here.")
|
||||
|
||||
md_updated, nb_updated = canonicalize_image_links_in_translations(
|
||||
translations_dir=translations_dir, image_dir=image_dir
|
||||
)
|
||||
|
||||
assert (md_updated, nb_updated) == (0, 0)
|
||||
|
||||
|
||||
def test_windows_separators_are_handled(tmp_path: Path):
|
||||
translations_dir = tmp_path / "translations"
|
||||
image_dir = tmp_path / "images_out"
|
||||
|
||||
md = translations_dir / "de" / "doc.md"
|
||||
original = (
|
||||
"\n"
|
||||
"\n"
|
||||
f"\n"
|
||||
)
|
||||
write_text(md, original)
|
||||
|
||||
md_updated, nb_updated = canonicalize_image_links_in_translations(
|
||||
translations_dir=translations_dir, image_dir=image_dir
|
||||
)
|
||||
|
||||
assert md_updated == 1
|
||||
assert nb_updated == 0
|
||||
|
||||
updated = md.read_text(encoding="utf-8")
|
||||
assert "translated_images\\zh-TW\\foo.webp" in updated
|
||||
assert "translated_images_fast\\pt-BR\\bar.webp" in updated
|
||||
assert f"{image_dir.name}\\zh-CN\\baz.webp" in updated
|
||||
@@ -0,0 +1,38 @@
|
||||
import pytest
|
||||
|
||||
from co_op_translator.utils.common.lang_utils import (
|
||||
normalize_language_code,
|
||||
normalize_language_codes,
|
||||
ALIAS_TO_BCP47,
|
||||
)
|
||||
|
||||
|
||||
def test_normalize_language_code_aliases():
|
||||
assert normalize_language_code("tw") == "zh-TW"
|
||||
assert normalize_language_code("cn") == "zh-CN"
|
||||
assert normalize_language_code("br") == "pt-BR"
|
||||
assert normalize_language_code("jp") == "ja"
|
||||
assert normalize_language_code("kr") == "ko"
|
||||
# generic zh maps to zh-CN per policy
|
||||
assert normalize_language_code("zh") == "zh-CN"
|
||||
|
||||
|
||||
def test_normalize_language_code_casing():
|
||||
assert normalize_language_code("pt-br") == "pt-BR"
|
||||
assert normalize_language_code("PT-br") == "pt-BR"
|
||||
assert normalize_language_code("zh-tw") == "zh-TW"
|
||||
assert normalize_language_code("EN-us") == "en-US"
|
||||
assert normalize_language_code("ja") == "ja"
|
||||
|
||||
|
||||
def test_normalize_language_codes_dedup_order():
|
||||
items = ["tw", "zh-TW", "cn", "ZH-cn", "br", "pt-br", "jp", "kr"]
|
||||
# Expect deduped canonical order of first occurrences
|
||||
out = normalize_language_codes(items)
|
||||
assert out == ["zh-TW", "zh-CN", "pt-BR", "ja", "ko"]
|
||||
|
||||
|
||||
def test_alias_table_contains_expected():
|
||||
# minimal sanity of important aliases
|
||||
for k in ["tw", "cn", "br", "jp", "kr", "zh"]:
|
||||
assert k in ALIAS_TO_BCP47
|
||||
Reference in New Issue
Block a user