Core: Add glossary support to programmatic API (#399)

This commit is contained in:
Minseok Song
2026-04-20 13:24:06 +09:00
committed by GitHub
parent 218ababe85
commit 88c6e0f13f
3 changed files with 145 additions and 66 deletions
+72 -65
View File
@@ -13,6 +13,7 @@ from co_op_translator.config.llm_config.config import LLMConfig
from co_op_translator.config.vision_config.config import VisionConfig
from co_op_translator.core.project.language_migrator import LanguageFolderMigrator
from co_op_translator.core.project.project_translator import ProjectTranslator
from co_op_translator.glossary import glossary_terms_scope
from co_op_translator.utils.common.file_utils import (
render_updated_readme_languages_table,
render_updated_readme_other_courses,
@@ -67,6 +68,7 @@ def run_translation(
root_dirs: Iterable[str] | None = None,
groups: Iterable[tuple[str, str | None]] | None = None,
repo_url: str | None = None,
glossaries: Iterable[str] | None = None,
dry_run: bool = False,
) -> None:
"""Programmatic translation entrypoint mirroring the translate CLI options."""
@@ -289,7 +291,9 @@ def run_translation(
try:
if update_readme_other_courses(readme_path):
click.echo("✅ Updated README 'Other courses' section from template.")
click.echo(
"✅ Updated README 'Other courses' section from template."
)
except Exception as e: # pragma: no cover
logger.warning(f"Failed to update README 'Other courses': {e}")
@@ -449,80 +453,83 @@ def run_translation(
else:
os.environ["TQDM_DISABLE"] = previous
aggregate_template = {
"markdown": 0,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 0,
"words": 0,
}
translation_types_for_summary: list[str] = []
if markdown:
translation_types_for_summary.append("markdown")
if images:
translation_types_for_summary.append("images")
if notebook:
translation_types_for_summary.append("notebook")
if not translation_types_for_summary:
translation_types_for_summary = ["markdown", "notebook", "images"]
with glossary_terms_scope(glossaries):
aggregate_template = {
"markdown": 0,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 0,
"words": 0,
}
translation_types_for_summary: list[str] = []
if markdown:
translation_types_for_summary.append("markdown")
if images:
translation_types_for_summary.append("images")
if notebook:
translation_types_for_summary.append("notebook")
if not translation_types_for_summary:
translation_types_for_summary = ["markdown", "notebook", "images"]
execution_targets: list[tuple[str, str | None, str | None]] = []
if groups is not None:
for per_root, per_translations in list(groups):
per_translations_dir: str | None = per_translations
per_lang_subdir: str | None = None
if per_translations is not None:
base_part, suffix = _split_lang_placeholder(per_translations)
per_translations_dir = base_part or None
per_lang_subdir = suffix
execution_targets.append((per_root, per_translations_dir, per_lang_subdir))
elif root_dirs is not None:
for per_root in list(root_dirs):
execution_targets.append((per_root, translations_dir, None))
else:
execution_targets.append((root_dir, translations_dir, None))
execution_targets: list[tuple[str, str | None, str | None]] = []
if groups is not None:
for per_root, per_translations in list(groups):
per_translations_dir: str | None = per_translations
per_lang_subdir: str | None = None
if per_translations is not None:
base_part, suffix = _split_lang_placeholder(per_translations)
per_translations_dir = base_part or None
per_lang_subdir = suffix
execution_targets.append(
(per_root, per_translations_dir, per_lang_subdir)
)
elif root_dirs is not None:
for per_root in list(root_dirs):
execution_targets.append((per_root, translations_dir, None))
else:
execution_targets.append((root_dir, translations_dir, None))
aggregated_estimate = dict(aggregate_template)
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
group_estimate = _compute_estimate_for_group(
language_codes=language_codes,
root_dir=per_root,
update=update,
markdown=markdown,
images=images,
notebook=notebook,
add_disclaimer=add_disclaimer,
translations_dir=per_translations_dir,
image_dir=image_dir,
lang_subdir=per_lang_subdir,
repo_url=repo_url,
)
aggregated_estimate = _merge_estimates(aggregated_estimate, group_estimate)
_echo_estimate_summary(aggregated_estimate, translation_types_for_summary)
multi_group_mode = len(execution_targets) > 1
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
with _tqdm_disabled(multi_group_mode):
_run_single_group(
aggregated_estimate = dict(aggregate_template)
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
group_estimate = _compute_estimate_for_group(
language_codes=language_codes,
root_dir=per_root,
update=update,
images=images,
markdown=markdown,
images=images,
notebook=notebook,
debug=debug,
save_logs=save_logs,
yes=yes,
add_disclaimer=add_disclaimer,
translations_dir=per_translations_dir,
image_dir=image_dir,
lang_subdir=per_lang_subdir,
repo_url=repo_url,
dry_run=dry_run,
)
aggregated_estimate = _merge_estimates(aggregated_estimate, group_estimate)
_echo_estimate_summary(aggregated_estimate, translation_types_for_summary)
multi_group_mode = len(execution_targets) > 1
for per_root, per_translations_dir, per_lang_subdir in execution_targets:
with _tqdm_disabled(multi_group_mode):
_run_single_group(
language_codes=language_codes,
root_dir=per_root,
update=update,
images=images,
markdown=markdown,
notebook=notebook,
debug=debug,
save_logs=save_logs,
yes=yes,
add_disclaimer=add_disclaimer,
translations_dir=per_translations_dir,
image_dir=image_dir,
lang_subdir=per_lang_subdir,
repo_url=repo_url,
dry_run=dry_run,
)
+16 -1
View File
@@ -1,4 +1,5 @@
from typing import Iterable
from contextlib import contextmanager
from typing import Iterable, Iterator
_glossary_terms: list[str] = []
@@ -8,6 +9,9 @@ def normalize_glossary_terms(glossary_terms: Iterable[str] | None) -> list[str]:
if not glossary_terms:
return []
if isinstance(glossary_terms, str):
glossary_terms = [glossary_terms]
normalized: list[str] = []
seen: set[str] = set()
for term in glossary_terms:
@@ -32,6 +36,17 @@ def get_glossary_terms() -> list[str]:
return list(_glossary_terms)
@contextmanager
def glossary_terms_scope(glossary_terms: Iterable[str] | None) -> Iterator[None]:
"""Temporarily set glossary terms for one translation API invocation."""
previous_terms = get_glossary_terms()
set_glossary_terms(glossary_terms)
try:
yield
finally:
set_glossary_terms(previous_terms)
def _build_glossary_lines() -> list[str]:
terms = get_glossary_terms()
if not terms:
+57
View File
@@ -3,6 +3,7 @@ from unittest.mock import MagicMock
import pytest
from co_op_translator.api import translation as api
from co_op_translator.glossary import get_glossary_terms, set_glossary_terms
@pytest.mark.asyncio
@@ -310,3 +311,59 @@ async def test_run_translation_dry_run_uses_virtual_readme_without_writing(tmp_p
assert readme_path.read_text(encoding="utf-8") == original_readme
api.update_readme_languages_table.assert_not_called()
api.update_readme_other_courses.assert_not_called()
@pytest.mark.asyncio
async def test_run_translation_accepts_glossaries_and_restores_previous_terms(tmp_path):
root_dir = tmp_path
api.Config.check_configuration = MagicMock(return_value=None)
api.LLMConfig.validate_connectivity = MagicMock(return_value=None)
api.setup_logging = MagicMock(return_value=None)
project_translator_instance = MagicMock()
project_translator_class = MagicMock(return_value=project_translator_instance)
api.ProjectTranslator = project_translator_class
observed_terms: list[list[str]] = []
def fake_estimate_tokens(*args, **kwargs):
observed_terms.append(get_glossary_terms())
return {
"markdown": 10,
"notebook": 0,
"images": 0,
"outdated_markdown": 0,
"outdated_notebook": 0,
"outdated_images": 0,
"outdated": 0,
"total": 10,
}
api.estimate_translation_tokens = MagicMock(side_effect=fake_estimate_tokens)
api.estimate_translation_words = MagicMock(
return_value={
"markdown": 5,
"notebook": 0,
"images": 0,
"outdated": 0,
"total": 5,
}
)
set_glossary_terms(["Existing Term"])
try:
api.run_translation(
language_codes="ko",
root_dir=str(root_dir),
markdown=True,
glossaries=[" Co-op Translator ", "Co-op Translator", "Azure AI"],
)
assert observed_terms == [["Co-op Translator", "Azure AI"]]
assert get_glossary_terms() == ["Existing Term"]
project_translator_instance.translate_project.assert_called_once_with(
update=False,
)
finally:
set_glossary_terms([])