Core, Build, Docs: Improve Translation Management & Metadata Handling (#76)

This commit is contained in:
Minseok Song
2025-02-01 02:19:20 +09:00
committed by GitHub
parent 6ee6b2163d
commit b98c1bab14
19 changed files with 1934 additions and 608 deletions
+4 -1
View File
@@ -417,4 +417,7 @@ dist
.pytest_cache/
# Egg info
co_op_translator.egg-info/
co_op_translator.egg-info/
# test repo
test_repo
+33
View File
@@ -56,6 +56,39 @@ poetry shell
poetry install
```
### Manual testing
Before submitting a PR, it's important to test the translation functionality with real documentation:
1. Create a test directory in the root directory:
```bash
mkdir test_docs
```
2. Copy some markdown documentation and images you want to translate into the test directory. For example:
```bash
cp /path/to/your/docs/*.md test_docs/
cp /path/to/your/images/*.png test_docs/
```
3. Install the package locally:
```bash
pip install -e .
```
4. Run Co-op Translator on your test documents:
```bash
python -m co_op_translator --language-codes ko --root-dir test_docs
```
5. Check the translated files in `test_docs/translations` and `test_docs/translated_images` to verify:
- The translation quality
- The metadata comments are correct
- The original markdown structure is preserved
- Links and images are working properly
This manual testing helps ensure that your changes work well in real-world scenarios.
### Environment variables
1. Create an `.env` file in the root directory by copying the provided `.env.template` file.
+2 -1
View File
@@ -1,6 +1,6 @@
[tool.poetry]
name = "co_op_translator"
version = "0.6.3"
version = "0.7.0b1"
description = "Easily automate multilingual translations for your projects with co-op-translator, powered by advanced LLM technology."
authors = [
"Minseok Song <skytin1004@gmail.com>",
@@ -95,6 +95,7 @@ nest-asyncio = "^1.6.0"
debugpy = "^1.8.1"
stack-data = "^0.6.3"
Pygments = "^2.18.0"
freezegun = "^1.5.1"
[tool.pytest.ini_options]
minversion = "6.0"
+1
View File
@@ -1,3 +1,4 @@
[pytest]
testpaths = tests
pythonpath = src
asyncio_mode = auto
+1
View File
@@ -71,3 +71,4 @@ nest-asyncio~=1.6.0
debugpy~=1.8.1
stack-data~=0.6.3
Pygments~=2.18.0
freezegun~=1.5.1
+130 -109
View File
@@ -6,6 +6,8 @@ import yaml
from co_op_translator.core.project.project_translator import ProjectTranslator
from co_op_translator.config.base_config import Config
from co_op_translator.config.vision_config.config import VisionConfig
import os
from pathlib import Path
logger = logging.getLogger(__name__)
@@ -23,13 +25,6 @@ logger = logging.getLogger(__name__)
default=".",
help="Root directory of the project (default is current directory).",
)
@click.option(
"--add",
"-a",
is_flag=True,
default=True,
help="Add new translations without deleting existing ones (default behavior).",
)
@click.option(
"--update",
"-u",
@@ -45,7 +40,7 @@ logger = logging.getLogger(__name__)
is_flag=True,
help="Check translated files for errors and retry translation if needed.",
)
def main(language_codes, root_dir, add, update, images, markdown, debug, check):
def main(language_codes, root_dir, update, images, markdown, debug, check):
"""
CLI for translating project files.
@@ -56,7 +51,7 @@ def main(language_codes, root_dir, add, update, images, markdown, debug, check):
translate -l "es fr de" -r "./my_project"
2. Add only new Korean image translations (no existing translations are deleted):
translate -l "ko" -img -a
translate -l "ko" -img
3. Update all Korean translations (Warning: This deletes all existing Korean translations before re-translating):
translate -l "ko" -u
@@ -65,7 +60,7 @@ def main(language_codes, root_dir, add, update, images, markdown, debug, check):
translate -l "ko" -img -u
5. Add new markdown translations for Korean without affecting other translations:
translate -l "ko" -md -a
translate -l "ko" -md
6. Check translated files for errors and retry translations if necessary:
translate -l "ko" -chk
@@ -80,111 +75,137 @@ def main(language_codes, root_dir, add, update, images, markdown, debug, check):
- translate -l "ko" -d: Enable debug logging.
"""
# Check that the required environment variables are set
Config.check_configuration()
try:
# Check that the required environment variables are set
Config.check_configuration()
# Check Computer Vision availability and set markdown mode
cv_available = VisionConfig.check_configuration()
# Initialize translation mode
cv_available = VisionConfig.check_configuration()
# Set markdown-only mode if Computer Vision is not available
if not cv_available:
markdown = True
click.echo(
"Computer Vision is not configured: Automatically switching to markdown-only mode."
)
click.echo(
"To enable image translation, please add Computer Vision credentials to your environment variables."
)
click.echo("See the .env.template file for required variables.")
elif markdown:
click.echo("Starting in markdown-only mode: Image translation is disabled.")
click.echo(
"To enable image translation, run without the -md flag and ensure Computer Vision is configured."
)
if debug:
logging.basicConfig(level=logging.DEBUG)
logging.debug("Debug mode enabled.")
else:
logging.basicConfig(level=logging.CRITICAL)
# Show warning if 'all' is selected
if language_codes == "all":
click.echo(
"Warning: Translating all languages at once can take a significant amount of time, especially when dealing with large markdown-based open-source projects that have many documents."
)
click.echo(
"For better efficiency, it's recommended that contributors handle individual languages and upload their translations separately."
)
# Option to proceed or not
confirmation_all = click.prompt(
"Do you still want to proceed with translating all languages? Type 'yes' to continue",
type=str,
)
if confirmation_all.lower() != "yes":
click.echo("Translation for 'all' languages cancelled.")
return
else:
click.echo("Proceeding with translation for all languages...")
# Show warning and prompt if update is selected
if update:
click.echo(
f"Warning: The update command will delete all existing translations for '{language_codes}' and re-translate everything."
)
confirmation_update = click.prompt(
"Do you want to continue? Type 'yes' to proceed", type=str
)
if confirmation_update.lower() != "yes":
click.echo("Update cancelled by user.")
return
else:
click.echo("Proceeding with update...")
# Language code parsing logic
if language_codes == "all":
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file)
language_codes = " ".join(
[
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
)
logging.debug(
f"Loaded language codes from font mapping: {language_codes}"
)
# Initialize ProjectTranslator with markdown_only flag
translator = ProjectTranslator(language_codes, root_dir, markdown_only=markdown)
if check:
# Call check_and_retry_translations if --check is passed
click.echo(f"Checking translated files for errors in {language_codes}...")
asyncio.run(translator.check_and_retry_translations())
else:
# Set default values for images and markdown flags
# Determine translation mode based on flags and CV availability
if not images and not markdown:
markdown = True # Always translate markdown by default
images = True # Try to translate images by default
# If CV is not available, disable image translation
if not cv_available:
click.echo(
"Skipping image translation due to missing Computer Vision configuration"
)
# Default: translate both if possible
markdown = True
images = cv_available
elif images and not cv_available:
# User requested images but CV not available
images = False
click.echo(
"Computer Vision is not configured: Image translation will be disabled."
)
click.echo(
"To enable image translation, please add Computer Vision credentials to your environment variables."
)
click.echo("See the .env.template file for required variables.")
# Call translate_project
translator.translate_project(images=images, markdown=markdown, update=update)
# Log selected translation mode
mode_msg = "Translation mode: "
if markdown and images:
mode_msg += "markdown and images"
elif markdown:
mode_msg += "markdown only"
elif images:
mode_msg += "images only"
click.echo(mode_msg)
logger.info(f"Project translation completed for languages: {language_codes}")
if debug:
logging.basicConfig(level=logging.DEBUG)
logging.debug("Debug mode enabled.")
else:
logging.basicConfig(level=logging.CRITICAL)
# Validate root directory
root_path = Path(root_dir).resolve()
if not root_path.exists():
raise click.ClickException(f"Root directory does not exist: {root_dir}")
if not root_path.is_dir():
raise click.ClickException(f"Root path is not a directory: {root_dir}")
if not os.access(root_path, os.R_OK | os.W_OK):
raise click.ClickException(
f"Insufficient permissions for directory: {root_dir}"
)
# Show warning if 'all' is selected
if language_codes == "all":
click.echo(
"Warning: Translating all languages at once can take a significant amount of time, especially when dealing with large markdown-based open-source projects that have many documents."
)
click.echo(
"For better efficiency, it's recommended that contributors handle individual languages and upload their translations separately."
)
# Option to proceed or not
confirmation_all = click.prompt(
"Do you still want to proceed with translating all languages? Type 'yes' to continue",
type=str,
)
if confirmation_all.lower() != "yes":
click.echo("Translation for 'all' languages cancelled.")
return
else:
click.echo("Proceeding with translation for all languages...")
try:
with importlib.resources.path(
"co_op_translator.fonts", "font_language_mappings.yml"
) as mappings_path:
with open(mappings_path, "r", encoding="utf-8") as file:
font_mappings = yaml.safe_load(file)
if not font_mappings:
raise click.ClickException("Empty font mappings file")
language_codes = " ".join(
[
lang_code
for lang_code in font_mappings
if isinstance(font_mappings[lang_code], dict)
]
)
if not language_codes:
raise click.ClickException(
"No valid language codes found in font mappings"
)
logging.debug(
f"Loaded language codes from font mapping: {language_codes}"
)
except (FileNotFoundError, yaml.YAMLError) as e:
raise click.ClickException(f"Failed to load font mappings: {str(e)}")
# Show warning and prompt if update is selected
if update:
click.echo(
f"Warning: The update command will delete all existing translations for '{language_codes}' and re-translate everything."
)
confirmation_update = click.prompt(
"Do you want to continue? Type 'yes' to proceed", type=str
)
if confirmation_update.lower() != "yes":
click.echo("Update cancelled by user.")
return
else:
click.echo("Proceeding with update...")
# Initialize ProjectTranslator with determined settings
translator = ProjectTranslator(
language_codes, root_dir, markdown_only=markdown and not images
)
if check:
# Call check_and_retry_translations if --check is passed
click.echo(f"Checking translated files for errors in {language_codes}...")
asyncio.run(translator.check_and_retry_translations())
else:
# Call translate_project with determined settings
translator.translate_project(
images=images, markdown=markdown, update=update
)
logger.info(f"Project translation completed for languages: {language_codes}")
except Exception as e:
if debug:
logger.exception("An error occurred during translation")
raise click.ClickException(str(e))
if __name__ == "__main__":
+4
View File
@@ -17,3 +17,7 @@ EXCLUDED_DIRS = {
".devcontainer",
".pytest_cache",
}
# Maximum allowed difference in line breaks between original and translated text
# A margin is needed to account for added disclaimer and metadata
LINE_BREAK_MARGIN = 15
@@ -14,6 +14,20 @@ from co_op_translator.utils.llm.markdown_utils import (
)
from co_op_translator.config.font_config import FontConfig
from co_op_translator.config.llm_config.config import LLMConfig
from co_op_translator.utils.common.metadata_utils import (
calculate_file_hash,
create_metadata,
format_metadata_comment,
)
from co_op_translator.utils.llm.markdown_utils import (
process_markdown,
update_links,
generate_prompt_template,
count_links_in_markdown,
process_markdown_with_many_links,
replace_code_blocks_and_inline_code,
restore_code_blocks_and_inline_code,
)
logger = logging.getLogger(__name__)
@@ -31,6 +45,46 @@ class MarkdownTranslator(ABC):
self.root_dir = root_dir
self.font_config = FontConfig()
def calculate_file_hash(self, file_path: Path) -> str:
"""
Calculate MD5 hash of a file.
Args:
file_path (Path): Path to the file to calculate hash for.
Returns:
str: MD5 hash of the file content.
"""
return calculate_file_hash(file_path)
def create_metadata(self, original_file: Path, language_code: str) -> dict:
"""
Create metadata for a translated file.
Args:
original_file (Path): Path to the original file
language_code (str): Target language code
Returns:
dict: Metadata dictionary containing file information
"""
return create_metadata(original_file, language_code, self.root_dir)
def format_metadata_comment(self, metadata: dict) -> str:
"""
Convert a metadata dictionary into a formatted HTML comment.
This function serializes the metadata dictionary as a JSON string with indentation,
wraps it in a custom HTML comment format, and returns the resulting string.
Args:
metadata (dict): A dictionary containing metadata to be formatted.
Returns:
str: A string containing the metadata formatted as an HTML comment.
"""
return format_metadata_comment(metadata)
async def translate_markdown(
self,
document: str,
@@ -48,10 +102,14 @@ class MarkdownTranslator(ABC):
markdown_only (bool): Whether we're in markdown-only mode.
Returns:
str: The translated content with updated links and a disclaimer appended.
str: The translated content with metadata, updated links and a disclaimer.
"""
md_file_path = Path(md_file_path)
# Create and format metadata
metadata = self.create_metadata(md_file_path, language_code)
metadata_comment = self.format_metadata_comment(metadata)
# Step 1: Replace code blocks and inline code with placeholders
document_with_placeholders, placeholder_map = (
replace_code_blocks_and_inline_code(document)
@@ -96,7 +154,7 @@ class MarkdownTranslator(ABC):
markdown_only=markdown_only,
)
disclaimer = await self.generate_disclaimer(language_code)
updated_content += "\n\n" + disclaimer
updated_content = metadata_comment + updated_content + "\n\n" + disclaimer
return updated_content
@@ -0,0 +1,275 @@
from pathlib import Path
import logging
import json
from co_op_translator.utils.common.file_utils import get_unique_id
logger = logging.getLogger(__name__)
class DirectoryManager:
"""
Manages directory structure and file cleanup for translation project.
"""
def __init__(
self,
root_dir: Path,
translations_dir: Path,
language_codes: list[str],
excluded_dirs: list[str],
):
"""
Initialize DirectoryManager.
Args:
root_dir: Root directory containing original files
translations_dir: Directory for translated files
language_codes: List of target language codes
excluded_dirs: List of directories to exclude
"""
self.root_dir = root_dir
self.translations_dir = translations_dir
self.language_codes = language_codes
self.excluded_dirs = excluded_dirs
def sync_directory_structure(
self, markdown: bool = True, images: bool = True
) -> tuple[int, int, int]:
"""
Synchronize the directory structure of translations with the original structure.
Process:
1. Scan original directory structure
2. For each language:
- Create missing directories that exist in original
- Remove directories that don't exist in original
Args:
markdown: Whether to sync markdown directories
images: Whether to sync image directories
Returns:
tuple[int, int, int]: (created_dirs, removed_dirs, synced_langs)
"""
created_count = 0
removed_count = 0
# Get original directory structure (excluding files)
original_dirs = set()
for path in self.root_dir.rglob("*"):
if path.is_dir() and not any(
excluded in str(path) for excluded in self.excluded_dirs
):
# For image-only mode, only include image directories
if (
not markdown
and not any(path.glob("*.png"))
and not any(path.glob("*.jpg"))
):
continue
# For markdown-only mode, only include markdown directories
if not images and not any(path.glob("*.md")):
continue
# Store relative path for comparison
original_dirs.add(path.relative_to(self.root_dir))
# Sync each language directory
for lang_code in self.language_codes:
lang_dir = self.translations_dir / lang_code
if not lang_dir.exists():
lang_dir.mkdir(parents=True)
logger.info(f"Created language directory: {lang_dir}")
# Get existing translation directories
translation_dirs = set()
if lang_dir.exists():
for path in lang_dir.rglob("*"):
if path.is_dir():
try:
translation_dirs.add(path.relative_to(lang_dir))
except ValueError:
continue
# Create missing directories
for orig_dir in original_dirs:
target_dir = lang_dir / orig_dir
if not target_dir.exists():
target_dir.mkdir(parents=True, exist_ok=True)
created_count += 1
logger.info(f"Created directory: {target_dir}")
# Remove extra directories that don't exist in original
for trans_dir in sorted(
translation_dirs, reverse=True
): # Sort reverse to handle deep paths first
if trans_dir not in original_dirs:
target_dir = lang_dir / trans_dir
try:
# Only remove if empty or contains no relevant files
has_relevant_files = False
if markdown and any(target_dir.rglob("*.md")):
has_relevant_files = True
if images and (
any(target_dir.rglob("*.png"))
or any(target_dir.rglob("*.jpg"))
):
has_relevant_files = True
if not has_relevant_files:
target_dir.rmdir() # This will only remove empty directories
removed_count += 1
logger.info(f"Removed empty directory: {target_dir}")
else:
logger.info(f"Skipping non-empty directory: {target_dir}")
except OSError as e:
logger.warning(f"Could not remove directory {target_dir}: {e}")
return created_count, removed_count, len(self.language_codes)
def cleanup_orphaned_translations(
self, markdown: bool = True, images: bool = True
) -> int:
"""
Remove translation files whose original files no longer exist.
Args:
markdown: Whether to clean up markdown files
images: Whether to clean up image files
Returns:
int: Number of removed translation files
"""
removed_count = 0
# Handle markdown files
if markdown:
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
if not translation_dir.exists():
continue
for trans_file in translation_dir.rglob("*.md"):
try:
# Read translation file and find metadata comment
content = trans_file.read_text(encoding="utf-8")
metadata_start = content.find("<!--")
if metadata_start == -1:
continue
metadata_end = content.find("-->", metadata_start)
if metadata_end == -1:
continue
metadata_str = content[
metadata_start + 4 : metadata_end
].strip()
metadata = json.loads(metadata_str)
# Check if original file exists
source_file = metadata.get("source_file")
if not source_file:
continue
original_file = self.root_dir / source_file
if not original_file.exists():
trans_file.unlink()
removed_count += 1
logger.debug(f"Removed orphaned translation: {trans_file}")
# Try to remove empty parent directories
parent = trans_file.parent
while parent != translation_dir:
if not any(parent.iterdir()): # If directory is empty
try:
parent.rmdir()
logger.debug(
f"Removed empty directory: {parent}"
)
except OSError:
break # Stop if directory is not empty or can't be removed
else:
break # Stop if directory is not empty
parent = parent.parent
except (json.JSONDecodeError, KeyError, FileNotFoundError) as e:
logger.warning(f"Error processing {trans_file}: {e}")
continue
# Handle image files
if images:
# Collect all image files in the original directory
original_images = {} # path_hash -> original_file_path
for image_file in self.root_dir.rglob("*"):
if not image_file.is_file():
continue
if image_file.suffix.lower() not in [".png", ".jpg", ".jpeg", ".gif"]:
continue
try:
path_hash = get_unique_id(image_file, self.root_dir)
original_images[path_hash] = image_file
except ValueError:
continue
# Check translated images in each language directory
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
if not translation_dir.exists():
continue
for image_file in translation_dir.rglob("*"):
if not image_file.is_file():
continue
if image_file.suffix.lower() not in [
".png",
".jpg",
".jpeg",
".gif",
]:
continue
# Image files follow the pattern: [original_name].[path_hash].[lang_code].[ext]
try:
# Split filename into parts
parts = image_file.name.split(".")
if (
len(parts) < 4
): # Need at least name, hash, lang_code, and extension
continue
# Last part is extension, second to last is lang_code, third to last is path_hash
extension = parts[-1]
lang_code = parts[-2]
path_hash = parts[-3]
if lang_code not in self.language_codes:
continue
# Check if original file with matching path hash exists
if path_hash not in original_images:
image_file.unlink()
removed_count += 1
logger.debug(f"Removed orphaned image: {image_file}")
# Try to remove empty parent directories
parent = image_file.parent
while parent != translation_dir:
if not any(parent.iterdir()): # If directory is empty
try:
parent.rmdir()
logger.debug(
f"Removed empty directory: {parent}"
)
except OSError:
break # Stop if directory is not empty or can't be removed
else:
break # Stop if directory is not empty
parent = parent.parent
except Exception as e:
logger.warning(f"Error processing image {image_file}: {e}")
continue
return removed_count
@@ -1,8 +1,6 @@
import logging
import os
from pathlib import Path
import asyncio
from tqdm.asyncio import tqdm
from co_op_translator.core.llm import (
markdown_translator,
text_translator,
@@ -11,20 +9,12 @@ from co_op_translator.core.vision import (
image_translator,
)
from co_op_translator.config.constants import (
SUPPORTED_IMAGE_EXTENSIONS,
EXCLUDED_DIRS,
SUPPORTED_IMAGE_EXTENSIONS,
)
from co_op_translator.utils.common.file_utils import (
read_input_file,
handle_empty_document,
get_filename_and_extension,
filter_files,
generate_translated_filename,
delete_translated_images_by_language_code,
delete_translated_markdown_files_by_language_code,
)
from co_op_translator.utils.common.task_utils import worker
from co_op_translator.utils.llm.markdown_utils import compare_line_breaks
from .directory_manager import DirectoryManager
from .translation_manager import TranslationManager
logger = logging.getLogger(__name__)
@@ -61,357 +51,62 @@ class ProjectTranslator:
self.root_dir
)
async def translate_image(self, image_path, language_code):
"""
Translate an image and handle file permissions or path errors.
"""
image_path = Path(image_path).resolve()
if not self.image_translator:
logger.info(
f"Image translation skipped for {image_path} due to missing Computer Vision configuration"
)
return str(
image_path
) # Return original image path when translation is not available
if image_path.exists() and image_path.is_file():
logger.info(f"Image exists: {image_path}")
if os.access(image_path, os.R_OK):
logger.info(f"Read permission granted for: {image_path}")
else:
logger.warning(f"Read permission denied for: {image_path}")
else:
logger.error(f"Image does not exist or is not a valid file: {image_path}")
try:
translated_image_path = self.image_translator.translate_image(
image_path, language_code, self.image_dir
)
logger.info(
f"Translated image {image_path} to {language_code} and saved to {translated_image_path}"
)
except Exception as e:
logger.error(f"Failed to translate image {image_path}: {e}", exc_info=True)
async def translate_markdown(self, file_path, language_code):
"""
Translate a markdown file to the specified language.
Args:
file_path (Path): Path to the markdown file.
language_code (str): The target language code.
"""
file_path = Path(file_path).resolve()
try:
document = read_input_file(file_path)
if not document:
relative_path = file_path.relative_to(self.root_dir)
output_file = self.translations_dir / language_code / relative_path
handle_empty_document(file_path, output_file)
return
# First attempt at translation
translated_content = await self.markdown_translator.translate_markdown(
document, language_code, file_path, markdown_only=self.markdown_only
)
if not translated_content:
logger.error(
f"Translation failed for {file_path}: Empty translation result"
)
return
# Check if translation format is broken (e.g., line breaks mismatch)
if compare_line_breaks(document, translated_content):
logger.warning(f"Translation failed for {file_path}. Retrying...")
# Retry translation
translated_content = await self.markdown_translator.translate_markdown(
document, language_code, file_path, markdown_only=self.markdown_only
)
if not translated_content:
logger.error(
f"Retry translation failed for {file_path}: Empty translation result"
)
return
relative_path = file_path.relative_to(self.root_dir)
translated_path = self.translations_dir / language_code / relative_path
translated_path.parent.mkdir(parents=True, exist_ok=True)
try:
with open(translated_path, "w", encoding="utf-8") as f:
f.write(translated_content)
logger.info(
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
except Exception as e:
logger.error(f"Failed to write translation to {translated_path}: {e}")
except Exception as e:
logger.error(f"Failed to translate {file_path}: {e}")
async def translate_all_markdown_files(self, update=False):
"""
Translate all markdown files sequentially, with optional update mode to refresh translations.
"""
logger.info("Starting markdown translation tasks...")
# Step 1: If update is True, delete all existing translated markdown files
if update:
for language_code in self.language_codes:
delete_translated_markdown_files_by_language_code(
language_code, self.translations_dir
)
logger.info(
f"Deleted all translated markdown files for language: {language_code}"
)
# Step 2: Collect markdown files for translation
markdown_files = filter_files(self.root_dir, EXCLUDED_DIRS)
tasks = []
for md_file_path in markdown_files:
md_file_path = md_file_path.resolve()
if md_file_path.suffix == ".md":
for language_code in self.language_codes:
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_path = (
self.translations_dir / language_code / relative_path
)
if not update and translated_md_path.exists():
logger.info(
f"Skipping already translated markdown file: {translated_md_path}"
)
continue
logger.info(
f"Translating markdown file: {md_file_path} for language: {language_code}"
)
# Create a task for each markdown file translation
tasks.append(
lambda md_file_path=md_file_path, language_code=language_code: self.translate_markdown(
md_file_path, language_code
)
)
if tasks: # Check if there are tasks to process
# Step 3: Process markdown translations sequentially using the sequential API request queue
await self.process_api_requests_sequential(
tasks, "Translating markdown files"
)
else:
logger.warning("No markdown files found for translation.")
async def translate_all_image_files(self, update=False):
"""
Translate all image files, with optional update mode to refresh translations.
"""
logger.info("Starting image translation tasks...")
# Step 1: If update is True, delete all existing translated images
if update:
for language_code in self.language_codes:
delete_translated_images_by_language_code(language_code, self.image_dir)
logger.info(
f"Deleted all translated images for language: {language_code}"
)
# Step 2: Collect image files for translation
image_files = filter_files(self.root_dir, EXCLUDED_DIRS)
tasks = []
for image_file_path in image_files:
image_file_path = image_file_path.resolve()
if (
get_filename_and_extension(image_file_path)[1]
in SUPPORTED_IMAGE_EXTENSIONS
):
for language_code in self.language_codes:
translated_filename = generate_translated_filename(
image_file_path, language_code, self.root_dir
)
translated_image_path = Path(self.image_dir) / translated_filename
if not update and translated_image_path.exists():
logger.info(
f"Skipping already translated image: {translated_image_path}"
)
continue
logger.info(
f"Translating image: {image_file_path} for language: {language_code}"
)
tasks.append(self.translate_image(image_file_path, language_code))
# Step 3: Process image translations using API request queue
await self.process_api_requests_parallel(tasks, "Translating images")
async def translate_project_async(self, images=False, markdown=False, update=False):
"""
Translate the entire project, including both markdown and image files.
Args:
images (bool): Flag to indicate if images should be translated.
markdown (bool): Flag to indicate if markdown files should be translated.
update (bool): Flag to indicate if existing translations should be updated.
"""
logger.info("Starting project translation tasks...")
if not images and not markdown:
images = True
markdown = True
# Add tasks for image translation
if images:
await self.translate_all_image_files(update=update)
# Add tasks for markdown translation
if markdown:
await self.translate_all_markdown_files(update=update)
# Initialize managers
self.directory_manager = DirectoryManager(
self.root_dir, self.translations_dir, self.language_codes, EXCLUDED_DIRS
)
self.translation_manager = TranslationManager(
self.root_dir,
self.translations_dir,
self.image_dir,
self.language_codes,
EXCLUDED_DIRS,
SUPPORTED_IMAGE_EXTENSIONS,
self.markdown_translator,
self.image_translator,
self.markdown_only,
)
def translate_project(self, images=False, markdown=False, update=False):
"""
Public method to start the project translation.
Public synchronous method to start the project translation.
Args:
images (bool): Whether to translate images.
markdown (bool): Whether to translate markdown files.
update (bool): Whether to update existing translations.
images: Whether to translate images
markdown: Whether to translate markdown files
update: Whether to update existing translations
"""
asyncio.run(
self.translate_project_async(
self.translation_manager.translate_project_async(
images=images, markdown=markdown, update=update
)
)
async def check_and_retry_translations(self):
"""
Check translated files for errors and retry translation if needed.
Display a single progress bar for both checking and retry processes.
"""
total_files_checked = 0
files_to_translate = []
"""Check and retry translations for all files."""
# Check outdated files first
modified_count, errors = await self.translation_manager.check_outdated_files()
logger.info(f"Found {modified_count} outdated files")
if errors:
logger.warning(f"Errors during checking outdated files: {errors}")
# Collect all markdown files
markdown_files = [
file
for file in filter_files(self.root_dir, EXCLUDED_DIRS)
if file.suffix == ".md"
]
all_markdown_files = [
(file, language_code)
for file in markdown_files
for language_code in self.language_codes
]
# Then translate all files
markdown_count, markdown_errors = (
await self.translation_manager.translate_all_markdown_files()
)
logger.info(f"Translated {markdown_count} markdown files")
if markdown_errors:
logger.warning(f"Errors during markdown translation: {markdown_errors}")
total_files = len(all_markdown_files)
# Finally translate images if image translator is available
image_count, image_errors = (
await self.translation_manager.translate_all_image_files()
)
logger.info(f"Translated {image_count} image files")
if image_errors:
logger.warning(f"Errors during image translation: {image_errors}")
if total_files == 0:
logger.warning("No markdown files found for checking.")
return
logger.info("Checking translated files for errors...")
# Step 1: Check all markdown files and collect files that need translation
with tqdm(
total=total_files, desc="Checking files", unit="file"
) as progress_bar:
for md_file_path, language_code in all_markdown_files:
md_file_path = Path(md_file_path).resolve()
total_files_checked += 1
# Find the path of the translated file
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_file_path = (
self.translations_dir / language_code / relative_path
)
if not translated_md_file_path.exists():
logger.warning(
f"Translated file does not exist: {translated_md_file_path}"
)
files_to_translate.append((md_file_path, language_code))
progress_bar.update(1)
continue
# Read the content of both original and translated files
original_content = read_input_file(md_file_path)
translated_content = read_input_file(translated_md_file_path)
# Check if line breaks are mismatched
if compare_line_breaks(original_content, translated_content):
files_to_translate.append((md_file_path, language_code))
logger.warning(
f"Detected formatting issue in {translated_md_file_path}"
)
# Update the progress bar after each file is checked
progress_bar.update(1)
# Step 2: Translate missing or mismatched files
if files_to_translate:
logger.info(f"Starting translation for {len(files_to_translate)} files...")
# Create a progress bar for translations
with tqdm(
total=len(files_to_translate), desc="Translating files", unit="file"
) as translation_progress_bar:
for md_file_path, language_code in files_to_translate:
logger.info(f"Translating {md_file_path} to {language_code}...")
# Translate the file
await self.translate_markdown(md_file_path, language_code)
# Update the progress bar for translation process
translation_progress_bar.update(1)
logger.info(f"Total files translated: {len(files_to_translate)}")
else:
logger.info("No formatting issues found in the translated files.")
logger.info(f"Total files checked: {total_files_checked}")
async def process_api_requests_parallel(self, tasks, task_desc):
"""
Process API requests using a queue system for better resource management (Parallel).
"""
if not tasks: # No tasks to process
logger.warning("No tasks available for processing.")
return
task_queue = asyncio.Queue()
# Step 1: Populate the queue with tasks
for task in tasks:
task_queue.put_nowait(task)
# Step 2: Create a progress bar
with tqdm(total=len(tasks), desc=task_desc) as progress_bar:
# Step 3: Create worker tasks to process the queue
workers = [
asyncio.create_task(worker(task_queue, progress_bar)) for _ in range(5)
]
# Step 4: Wait until all tasks are processed
await task_queue.join()
# Ensure all workers have completed
for worker_task in workers:
worker_task.cancel()
async def process_api_requests_sequential(self, tasks, task_desc):
"""
Process API requests sequentially, one after another (Sequential).
"""
if not tasks: # No tasks to process
logger.warning("No tasks available for processing.")
return
total_tasks = len(tasks)
with tqdm(total=total_tasks, desc=task_desc) as progress_bar:
for task in tasks:
await task() # Execute each task sequentially
progress_bar.update(1) # Update progress bar
return (
modified_count + markdown_count + image_count,
errors + markdown_errors + image_errors,
)
@@ -0,0 +1,772 @@
from pathlib import Path
import logging
from typing import List
from tqdm import tqdm
import re
import json
import os
import asyncio
from co_op_translator.utils.common.file_utils import (
read_input_file,
filter_files,
delete_translated_images_by_language_code,
delete_translated_markdown_files_by_language_code,
get_filename_and_extension,
generate_translated_filename,
handle_empty_document,
)
from co_op_translator.utils.common.metadata_utils import calculate_file_hash
from co_op_translator.core.llm.markdown_translator import MarkdownTranslator
from co_op_translator.core.project.directory_manager import DirectoryManager
from co_op_translator.config.constants import SUPPORTED_IMAGE_EXTENSIONS
from co_op_translator.utils.common.task_utils import worker
from co_op_translator.utils.llm.markdown_utils import compare_line_breaks
logger = logging.getLogger(__name__)
class TranslationManager:
"""
Manages the translation of markdown and image files.
"""
def __init__(
self,
root_dir: Path,
translations_dir: Path,
image_dir: Path,
language_codes: list[str],
excluded_dirs: list[str],
supported_image_extensions: list[str],
markdown_translator: MarkdownTranslator,
image_translator=None,
markdown_only: bool = False,
):
"""
Initialize TranslationManager.
Args:
root_dir: Root directory containing original files
translations_dir: Directory for translated files
image_dir: Directory for translated images
language_codes: List of target language codes
excluded_dirs: List of directories to exclude
supported_image_extensions: List of supported image extensions
markdown_translator: Translator instance for markdown files
image_translator: Translator instance for image files
markdown_only: Whether to only translate markdown files
"""
self.root_dir = root_dir
self.translations_dir = translations_dir
self.image_dir = image_dir
self.language_codes = language_codes
self.excluded_dirs = excluded_dirs
self.supported_image_extensions = supported_image_extensions
self.markdown_translator = markdown_translator
self.image_translator = image_translator
self.markdown_only = markdown_only
self.directory_manager = DirectoryManager(
root_dir, translations_dir, language_codes, excluded_dirs
)
async def translate_image(self, image_path: Path, language_code: str) -> str:
"""
Translate an image and handle file permissions or path errors.
Args:
image_path (Path): Path to the image file
language_code (str): Target language code
Returns:
str: The translated image path if successful, otherwise the original path
"""
image_path = Path(image_path).resolve()
if not self.image_translator:
logger.info(
f"Image translation skipped for {image_path} due to missing Computer Vision configuration"
)
return str(
image_path
) # Return original image path when translation is not available
if image_path.exists() and image_path.is_file():
logger.info(f"Image exists: {image_path}")
if os.access(image_path, os.R_OK):
logger.info(f"Read permission granted for: {image_path}")
else:
logger.warning(f"Read permission denied for: {image_path}")
return str(image_path)
else:
logger.error(f"Image does not exist or is not a valid file: {image_path}")
return str(image_path)
try:
translated_image_path = self.image_translator.translate_image(
image_path, language_code, self.image_dir
)
logger.info(
f"Translated image {image_path} to {language_code} and saved to {translated_image_path}"
)
return str(translated_image_path)
except Exception as e:
logger.error(f"Failed to translate image {image_path}: {e}", exc_info=True)
return str(image_path)
async def translate_markdown(self, file_path: Path, language_code: str) -> str:
"""
Translate a markdown file to the specified language.
Args:
file_path (Path): Path to the markdown file.
language_code (str): The target language code.
Returns:
str: The path to translated markdown file if successful, otherwise empty string
"""
file_path = Path(file_path).resolve()
try:
document = read_input_file(file_path)
if not document:
relative_path = file_path.relative_to(self.root_dir)
output_file = self.translations_dir / language_code / relative_path
handle_empty_document(file_path, output_file)
return str(output_file)
# First attempt at translation
translated_content = await self.markdown_translator.translate_markdown(
document, language_code, file_path, markdown_only=self.markdown_only
)
if not translated_content:
logger.error(
f"Translation failed for {file_path}: Empty translation result"
)
return ""
# Check if translation format is broken (e.g., line breaks mismatch)
if compare_line_breaks(document, translated_content):
logger.warning(f"Translation failed for {file_path}. Retrying...")
# Retry translation
translated_content = await self.markdown_translator.translate_markdown(
document, language_code, file_path, markdown_only=self.markdown_only
)
if not translated_content:
logger.error(
f"Retry translation failed for {file_path}: Empty translation result"
)
return ""
relative_path = file_path.relative_to(self.root_dir)
translated_path = self.translations_dir / language_code / relative_path
translated_path.parent.mkdir(parents=True, exist_ok=True)
try:
with open(translated_path, "w", encoding="utf-8") as f:
f.write(translated_content)
logger.info(
f"Translated {file_path} to {language_code} and saved to {translated_path}"
)
return str(translated_path)
except Exception as e:
logger.error(f"Failed to write translation to {translated_path}: {e}")
return ""
except Exception as e:
logger.error(f"Failed to translate {file_path}: {e}")
return ""
async def translate_all_markdown_files(
self, update: bool = False
) -> tuple[int, list[str]]:
"""
Translate all markdown files in the project directory.
Args:
update (bool): If True, update existing translations. Defaults to False.
Returns:
tuple[int, list[str]]: A tuple containing:
- Number of files modified
- List of error messages
"""
modified_count = 0
errors = []
# Step 1: If update is True, delete all existing translated markdown files
if update:
for language_code in self.language_codes:
delete_translated_markdown_files_by_language_code(
language_code, self.translations_dir
)
logger.info(
f"Deleted all translated markdown files for language: {language_code}"
)
# Step 2: Collect markdown files for translation
markdown_files = filter_files(self.root_dir, self.excluded_dirs)
tasks = []
for md_file_path in markdown_files:
md_file_path = md_file_path.resolve()
if md_file_path.suffix == ".md":
for language_code in self.language_codes:
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_path = (
self.translations_dir / language_code / relative_path
)
if not update and translated_md_path.exists():
logger.info(
f"Skipping already translated markdown file: {translated_md_path}"
)
continue
logger.info(
f"Translating markdown file: {md_file_path} for language: {language_code}"
)
# Create a task for each markdown file translation
tasks.append(
lambda md_file_path=md_file_path, language_code=language_code: self.translate_markdown(
md_file_path, language_code
)
)
if tasks: # Check if there are tasks to process
# Step 3: Process markdown translations sequentially using the sequential API request queue
results = await self.process_api_requests_sequential(
tasks, "Translating markdown files"
)
modified_count = sum(
1 for r in results if r
) # Count successful translations
errors = [
f"Failed to translate {task.__name__}"
for task, result in zip(tasks, results)
if not result
]
else:
logger.warning("No markdown files found for translation.")
return modified_count, errors
async def translate_all_image_files(self, update=False) -> tuple[int, list[str]]:
"""
Translate all image files, with optional update mode to refresh translations.
Args:
update (bool): If True, update existing translations. Defaults to False.
Returns:
tuple[int, list[str]]: A tuple containing:
- Number of files modified
- List of error messages
"""
modified_count = 0
errors = []
logger.info("Starting image translation tasks...")
# Step 1: If update is True, delete all existing translated images
if update:
for language_code in self.language_codes:
delete_translated_images_by_language_code(language_code, self.image_dir)
logger.info(
f"Deleted all translated images for language: {language_code}"
)
# Step 2: Collect image files for translation
image_files = filter_files(self.root_dir, self.excluded_dirs)
tasks = []
for image_file_path in image_files:
image_file_path = image_file_path.resolve()
if (
get_filename_and_extension(image_file_path)[1]
in self.supported_image_extensions
):
for language_code in self.language_codes:
translated_filename = generate_translated_filename(
image_file_path, language_code, self.root_dir
)
translated_image_path = Path(self.image_dir) / translated_filename
if not update and translated_image_path.exists():
logger.info(
f"Skipping already translated image: {translated_image_path}"
)
continue
logger.info(
f"Translating image: {image_file_path} for language: {language_code}"
)
tasks.append(self.translate_image(image_file_path, language_code))
if tasks:
# Step 3: Process image translations using API request queue
results = await self.process_api_requests_parallel(
tasks, "Translating images"
)
modified_count = sum(
1 for r in results if r != str(image_file_path)
) # Count successful translations
errors = [
f"Failed to translate {task.__name__}"
for task, result in zip(tasks, results)
if result == str(image_file_path)
]
else:
logger.warning("No image files found for translation.")
return modified_count, errors
async def check_outdated_files(self, update: bool = False) -> tuple[int, List[str]]:
"""
Check for outdated translated files by comparing metadata hash values and retranslate if needed.
Args:
update (bool): Whether to update existing translations regardless of hash values
Returns:
tuple[int, List[str]]: Number of retranslated files and list of errors
"""
modified_count = 0
errors = []
# Find all markdown files in root directory
markdown_files = [
f
for f in filter_files(self.root_dir, self.excluded_dirs)
if f.suffix.lower() == ".md"
]
# Create progress bar
with tqdm(total=len(markdown_files) * len(self.language_codes)) as pbar:
# Process each markdown file
for md_file in markdown_files:
try:
# Calculate relative path from root
relative_path = md_file.relative_to(self.root_dir)
# Translate to each language
for lang_code in self.language_codes:
try:
# Calculate target path
target_path = (
self.translations_dir / lang_code / relative_path
)
# Skip if target doesn't exist
if not target_path.exists():
pbar.update(1)
continue
# Read target file content and extract metadata
target_content = target_path.read_text(encoding="utf-8")
# Create current metadata for comparison
current_metadata = self.markdown_translator.create_metadata(
md_file, lang_code
)
current_hash = current_metadata.get("file_hash")
# Extract metadata from the target file using metadata_utils
metadata_match = re.search(
r"<!--\s*translation-metadata\s*(.*?)\s*-->",
target_content,
re.DOTALL,
)
if not metadata_match:
logger.warning(f"No metadata found in {target_path}")
pbar.update(1)
continue
try:
stored_metadata = json.loads(metadata_match.group(1))
except json.JSONDecodeError:
logger.warning(f"Invalid metadata in {target_path}")
pbar.update(1)
continue
# Get stored hash from metadata
stored_hash = stored_metadata.get("file_hash")
if not stored_hash:
logger.warning(
f"No file hash in metadata of {target_path}"
)
pbar.update(1)
continue
# Compare hashes and retranslate if different
if update or stored_hash != current_hash:
# Read original content
content = md_file.read_text(encoding="utf-8")
# Translate content
translated_content = (
await self.markdown_translator.translate_markdown(
content, lang_code, md_file, self.markdown_only
)
)
# Write translated content
target_path.write_text(
translated_content, encoding="utf-8"
)
modified_count += 1
logger.info(
f"Retranslated {md_file} to {lang_code} due to content changes"
)
except Exception as e:
error_msg = f"Error checking/retranslating {md_file} to {lang_code}: {str(e)}"
logger.error(error_msg)
errors.append(error_msg)
finally:
pbar.update(1)
except Exception as e:
error_msg = f"Error processing {md_file}: {str(e)}"
logger.error(error_msg)
errors.append(error_msg)
pbar.update(len(self.language_codes))
return modified_count, errors
async def translate_project_async(
self, images: bool = False, markdown: bool = False, update: bool = False
) -> tuple[int, list[str]]:
"""
Asynchronously translate the project.
The translation process follows these steps:
1. Remove orphaned files
2. Synchronize directory structure with source
3. Identify outdated translations
4. Perform translation on required files
Args:
images (bool): Whether to translate images. Defaults to False.
markdown (bool): Whether to translate markdown files. Defaults to False.
update (bool): Whether to update existing translations. Defaults to False.
Returns:
tuple[int, list[str]]: A tuple containing:
- Total number of files modified
- List of error messages
"""
logger.info("Starting project translation...")
total_modified = 0
all_errors = []
try:
# 1. Remove orphaned files first
logger.info("Removing orphaned files...")
with tqdm(total=1, desc="Cleaning orphaned files") as cleanup_progress:
removed_count = self.directory_manager.cleanup_orphaned_translations(
markdown=markdown, images=images
)
cleanup_progress.set_postfix_str(
"None" if removed_count == 0 else f"Removed: {removed_count}"
)
cleanup_progress.update(1)
# 2. Then synchronize directory structure
logger.info("Synchronizing directory structure...")
with tqdm(total=1, desc="Synchronizing directories") as sync_progress:
created, removed, _ = self.directory_manager.sync_directory_structure()
sync_progress.set_postfix_str(
"None"
if (created == 0 and removed == 0)
else f"Created: {created}, Removed: {removed}"
)
sync_progress.update(1)
# 3. Identify files requiring updates
if markdown:
with tqdm(total=1, desc="Checking translations") as check_progress:
outdated_files = self.get_outdated_translations()
check_progress.set_postfix_str(
"None"
if not outdated_files
else f"Found: {len(outdated_files)}"
)
check_progress.update(1)
if outdated_files:
with tqdm(
total=len(outdated_files), desc="Retranslating outdated files"
) as retrans_progress:
await self.retranslate_outdated_files(outdated_files)
retrans_progress.set_postfix_str(
f"Completed: {len(outdated_files)}"
)
retrans_progress.update(1)
# 4. Perform translation
if markdown:
md_modified, md_errors = await self.translate_all_markdown_files(
update=update
)
total_modified += md_modified
all_errors.extend(md_errors)
if images and not self.markdown_only:
img_modified, img_errors = await self.translate_all_image_files(
update=update
)
total_modified += img_modified
all_errors.extend(img_errors)
except Exception as e:
logger.error(f"Error during translation: {e}")
all_errors.append(str(e))
logger.info(f"Translation completed. Modified {total_modified} files.")
if all_errors:
logger.warning(f"Encountered {len(all_errors)} errors during translation")
return total_modified, all_errors
def get_outdated_translations(self) -> List[tuple[Path, Path]]:
"""
Identify translations that need updates by comparing metadata.
Returns:
List[tuple[Path, Path]]: List of (original file, translation file) tuples that need updates
"""
outdated_files = []
all_translation_files = []
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
if not translation_dir.exists():
continue
all_translation_files.extend(list(translation_dir.rglob("*.md")))
if not all_translation_files:
return []
for trans_file in all_translation_files:
lang_code = trans_file.parent.name
relative_path = trans_file.relative_to(self.translations_dir / lang_code)
original_file = self.root_dir / relative_path
if not original_file.exists():
continue
# Compare metadata
if self._is_translation_outdated(original_file, trans_file):
outdated_files.append((original_file, trans_file))
return outdated_files
async def retranslate_outdated_files(
self, outdated_files: List[tuple[Path, Path]]
) -> None:
"""
Retranslate the given outdated files.
Args:
outdated_files: List of (original_file, translation_file) tuples to retranslate
"""
if not outdated_files:
return
files_to_translate = []
for original_file, translation_file in outdated_files:
lang_code = translation_file.parent.name
files_to_translate.append((original_file, lang_code))
with tqdm(total=len(files_to_translate), desc="Retranslating") as progress:
for original_file, language_code in files_to_translate:
await self.translate_markdown(original_file, language_code)
progress.update(1)
progress.set_postfix_str(f"Current: {original_file.name}")
async def check_and_retry_translations(self):
"""
Check translated files for errors and retry translation if needed.
Display a single progress bar for both checking and retry processes.
"""
total_files_checked = 0
files_to_translate = []
# Collect all markdown files
markdown_files = [
file
for file in filter_files(self.root_dir, self.excluded_dirs)
if file.suffix == ".md"
]
all_markdown_files = [
(file, language_code)
for file in markdown_files
for language_code in self.language_codes
]
total_files = len(all_markdown_files)
if total_files == 0:
logger.warning("No markdown files found for checking.")
return
logger.info("Checking translated files for errors...")
# Step 1: Check all markdown files and collect files that need translation
with tqdm(
total=total_files, desc="Checking files", unit="file"
) as progress_bar:
for md_file_path, language_code in all_markdown_files:
md_file_path = Path(md_file_path).resolve()
total_files_checked += 1
# Find the path of the translated file
relative_path = md_file_path.relative_to(self.root_dir)
translated_md_file_path = (
self.translations_dir / language_code / relative_path
)
if not translated_md_file_path.exists():
logger.warning(
f"Translated file does not exist: {translated_md_file_path}"
)
files_to_translate.append((md_file_path, language_code))
progress_bar.update(1)
continue
# Read the content of both original and translated files
original_content = read_input_file(md_file_path)
translated_content = read_input_file(translated_md_file_path)
# Check if line breaks are mismatched
if compare_line_breaks(original_content, translated_content):
files_to_translate.append((md_file_path, language_code))
logger.warning(
f"Detected formatting issue in {translated_md_file_path}"
)
# Update the progress bar after each file is checked
progress_bar.update(1)
# Step 2: Translate missing or mismatched files
if files_to_translate:
logger.info(f"Starting translation for {len(files_to_translate)} files...")
# Create a progress bar for translations
with tqdm(
total=len(files_to_translate), desc="Translating files", unit="file"
) as translation_progress_bar:
for md_file_path, language_code in files_to_translate:
logger.info(f"Translating {md_file_path} to {language_code}...")
# Translate the file
await self.translate_markdown(
file_path=md_file_path,
language_code=language_code,
)
# Update the progress bar for translation process
translation_progress_bar.update(1)
logger.info(f"Total files translated: {len(files_to_translate)}")
else:
logger.info("No formatting issues found in the translated files.")
logger.info(f"Total files checked: {total_files_checked}")
async def process_api_requests_parallel(self, tasks, task_desc) -> list:
"""
Process API requests using a queue system for better resource management (Parallel).
"""
if not tasks: # No tasks to process
logger.warning("No tasks available for processing.")
return []
task_queue = asyncio.Queue()
# Step 1: Populate the queue with tasks
for task in tasks:
task_queue.put_nowait(task)
# Step 2: Create a progress bar
with tqdm(total=len(tasks), desc=task_desc) as progress_bar:
# Step 3: Create worker tasks to process the queue
workers = [
asyncio.create_task(worker(task_queue, progress_bar)) for _ in range(5)
]
# Step 4: Wait until all tasks are processed
await task_queue.join()
# Get results from completed tasks
results = [task.result() for task in workers]
# Ensure all workers have completed
for worker_task in workers:
worker_task.cancel()
return results
async def process_api_requests_sequential(self, tasks, task_desc) -> list:
"""
Process API requests sequentially, one after another (Sequential).
"""
if not tasks: # No tasks to process
logger.warning("No tasks available for processing.")
return []
total_tasks = len(tasks)
results = []
with tqdm(total=total_tasks, desc=task_desc) as progress_bar:
for task in tasks:
result = await task() # Execute each task sequentially
results.append(result)
progress_bar.update(1) # Update progress bar
return results
def _is_translation_outdated(
self, original_file: Path, translation_file: Path
) -> bool:
"""
Check if a translation needs to be updated by comparing original file's hash with the hash in translation metadata.
Args:
original_file (Path): Path to the original file
translation_file (Path): Path to the translation file
Returns:
bool: True if the translation needs to be updated, False otherwise
"""
if not translation_file.exists():
return True
try:
# Read translation file and find metadata comment
content = translation_file.read_text(encoding="utf-8")
metadata_match = re.search(
r"<!--\s*CO_OP_TRANSLATOR_METADATA:\s*(.*?)\s*-->",
content,
re.DOTALL,
)
if not metadata_match:
return True
try:
metadata = json.loads(metadata_match.group(1))
except json.JSONDecodeError:
return True
# Compare original file's hash with metadata
original_hash = calculate_file_hash(original_file)
stored_hash = metadata.get("original_hash")
if not stored_hash:
return True
return stored_hash != original_hash
except Exception:
return True
@@ -0,0 +1,82 @@
"""
This module contains utility functions for handling file metadata and hashing operations.
"""
import hashlib
import json
from datetime import datetime
from pathlib import Path
def calculate_file_hash(file_path: Path) -> str:
"""
Calculate MD5 hash of a file.
Args:
file_path (Path): Path to the file to calculate hash for.
Returns:
str: MD5 hash of the file content.
"""
hasher = hashlib.md5()
with open(file_path, "rb") as f:
for chunk in iter(lambda: f.read(4096), b""):
hasher.update(chunk)
return hasher.hexdigest()
def create_metadata(
original_file: Path, language_code: str, root_dir: Path | None = None
) -> dict:
"""
Create metadata for a translated file.
Args:
original_file (Path): Path to the original file
language_code (str): Target language code
root_dir (Path, optional): Root directory for relative path calculation
Returns:
dict: Metadata dictionary containing file information
"""
utc_time = datetime.utcnow()
formatted_time = utc_time.strftime("%Y-%m-%dT%H:%M:%S+00:00")
return {
"original_hash": calculate_file_hash(original_file),
"translation_date": formatted_time,
"source_file": (
str(original_file.relative_to(root_dir)) if root_dir else str(original_file)
),
"language_code": language_code,
}
def format_metadata_comment(metadata: dict) -> str:
"""
Convert a metadata dictionary into a formatted HTML comment.
This function serializes the metadata dictionary as a JSON string with indentation,
wraps it in a custom HTML comment format, and returns the resulting string.
Args:
metadata (dict): A dictionary containing metadata to be formatted.
Returns:
str: A string containing the metadata formatted as an HTML comment.
Example:
<!--
CO_OP_TRANSLATOR_METADATA:
{
"original_hash": "sample_hash",
"translation_date": "2025-01-30T13:02:53+00:00",
"source_file": "test.md",
"language_code": "ko"
}
-->
Total lines: 9
"""
metadata_json = json.dumps(metadata, indent=2)
formatted_comment = f"<!--\nCO_OP_TRANSLATOR_METADATA:\n{metadata_json}\n-->\n"
return formatted_comment
@@ -10,7 +10,10 @@ import tiktoken
from pathlib import Path
from urllib.parse import urlparse
import logging
from co_op_translator.config.constants import SUPPORTED_IMAGE_EXTENSIONS
from co_op_translator.config.constants import (
SUPPORTED_IMAGE_EXTENSIONS,
LINE_BREAK_MARGIN,
)
from co_op_translator.utils.common.file_utils import (
generate_translated_filename,
get_actual_image_path,
@@ -448,9 +451,7 @@ def compare_line_breaks(original_text, translated_text):
original_line_breaks = original_text.count("\n")
translated_line_breaks = translated_text.count("\n")
if (
abs(original_line_breaks - translated_line_breaks) > 5
): # Allow a margin for disclaimer
if abs(original_line_breaks - translated_line_breaks) > LINE_BREAK_MARGIN:
return True
return False
@@ -1,6 +1,7 @@
import pytest
from unittest.mock import AsyncMock, patch
import re
from pathlib import Path
from co_op_translator.core.llm.markdown_translator import MarkdownTranslator
@@ -28,7 +29,7 @@ def real_markdown_translator(tmp_path):
@pytest.mark.asyncio
async def test_translate_markdown_partial_mock(real_markdown_translator):
async def test_translate_markdown_partial_mock(real_markdown_translator, tmp_path):
"""Test the translation logic using the real code for:
- replace_code_blocks_and_inline_code
- restore_code_blocks_and_inline_code
@@ -38,6 +39,9 @@ async def test_translate_markdown_partial_mock(real_markdown_translator):
This ensures we verify the actual internal workflow,
while controlling the translation 'results' in _run_prompt.
"""
# Create test file
test_file = tmp_path / "example.md"
test_file.write_text(TEST_MD_CONTENT)
async def fake_prompt(prompt, index, total):
# Return a translation that preserves markdown structure and placeholders
@@ -59,7 +63,7 @@ async def test_translate_markdown_partial_mock(real_markdown_translator):
# Execute the markdown translation
result = await real_markdown_translator.translate_markdown(
document=TEST_MD_CONTENT, language_code="es", md_file_path="example.md"
document=TEST_MD_CONTENT, language_code="es", md_file_path=test_file
)
# Verify that the code block content is still present in the final output
@@ -79,13 +83,17 @@ async def test_translate_markdown_partial_mock(real_markdown_translator):
@pytest.mark.asyncio
async def test_translate_markdown_full_integration(real_markdown_translator):
async def test_translate_markdown_full_integration(real_markdown_translator, tmp_path):
"""A full integration test that avoids mocking _run_prompt at all.
This only works if the abstract _run_prompt has a default implementation
or if real_markdown_translator is a fully realized subclass."""
# Create test file
test_file = tmp_path / "example_full.md"
test_file.write_text(TEST_MD_CONTENT)
try:
result = await real_markdown_translator.translate_markdown(
document=TEST_MD_CONTENT, language_code="es", md_file_path="example_full.md"
document=TEST_MD_CONTENT, language_code="es", md_file_path=test_file
)
except NotImplementedError:
pytest.skip(
@@ -0,0 +1,157 @@
"""
Tests for DirectoryManager class
"""
import pytest
import os
from co_op_translator.core.project.directory_manager import DirectoryManager
@pytest.fixture
def setup_test_dirs(tmp_path):
"""Set up test directory structure"""
# Create test directories
root_dir = tmp_path / "test_project"
translations_dir = root_dir / "translations"
root_dir.mkdir()
# Create some test files
(root_dir / "test.md").write_text("# Test content")
(root_dir / "img").mkdir()
(root_dir / "img/test.png").write_bytes(b"fake png content")
return root_dir
@pytest.fixture
def setup_manager(setup_test_dirs):
"""Create a DirectoryManager instance with test configuration"""
root_dir = setup_test_dirs
translations_dir = root_dir / "translations"
language_codes = ["ko", "ja"]
excluded_dirs = ["node_modules", ".git"]
return DirectoryManager(
root_dir=root_dir,
translations_dir=translations_dir,
language_codes=language_codes,
excluded_dirs=excluded_dirs,
)
class TestDirectoryManager:
def test_init_directories(self, setup_manager, setup_test_dirs):
"""Test directory initialization"""
manager = setup_manager
translations_dir = setup_test_dirs / "translations"
manager.sync_directory_structure()
# Check if translations directory is created
assert translations_dir.exists()
# Check if language-specific directories are created
for lang in ["ko", "ja"]:
assert (translations_dir / lang).exists()
def test_cleanup_orphaned_translations(self, setup_test_dirs):
"""Test cleanup of orphaned translations"""
root_dir = setup_test_dirs
language_codes = ["ko"]
excluded_dirs = ["node_modules", ".git"]
translations_dir = root_dir / "translations"
# Create translations directory and files
translations_dir.mkdir()
ko_dir = translations_dir / "ko"
ko_dir.mkdir()
# Create test files
test_file = ko_dir / "test.md"
test_file.write_text(
"""<!--
{
"source_file": "test.md"
}
-->
# Test Translation"""
)
orphaned_file = ko_dir / "orphaned.md"
orphaned_file.write_text(
"""<!--
{
"source_file": "nonexistent.md"
}
-->
# Orphaned Translation"""
)
# Create original file only for test.md
(root_dir / "test.md").write_text("# Original Test")
manager = DirectoryManager(
root_dir=root_dir,
translations_dir=translations_dir,
language_codes=language_codes,
excluded_dirs=excluded_dirs,
)
# Run cleanup
manager.cleanup_orphaned_translations(markdown=True, images=False)
# Verify test.md translation still exists (has matching original)
assert test_file.exists()
# Verify orphaned.md was removed (no matching original)
assert not orphaned_file.exists()
def test_cleanup_orphaned_image_translations(self, setup_test_dirs):
"""Test cleanup of orphaned image translations"""
root_dir = setup_test_dirs
translations_dir = root_dir / "translations"
language_codes = ["ko"]
excluded_dirs = []
# Create original image
img_dir = root_dir / "img"
img_dir.mkdir(exist_ok=True)
original_img = img_dir / "test.png"
original_img.write_bytes(b"test image content")
# Create translated image with hash
translations_dir.mkdir(exist_ok=True)
ko_dir = translations_dir / "ko"
ko_dir.mkdir(exist_ok=True)
ko_img_dir = ko_dir / "img"
ko_img_dir.mkdir(exist_ok=True)
# Valid translated image (with correct hash)
from co_op_translator.utils.common.file_utils import get_unique_id
original_name, ext = os.path.splitext(original_img.name)
path_hash = get_unique_id(original_img, root_dir)
valid_trans_name = f"{original_name}.{path_hash}.ko{ext}"
valid_trans_img = ko_img_dir / valid_trans_name
valid_trans_img.write_bytes(b"translated content")
# Orphaned translated image (with incorrect hash)
orphaned_trans_img = ko_img_dir / "test.invalid_hash.ko.png"
orphaned_trans_img.write_bytes(b"orphaned content")
manager = DirectoryManager(
root_dir=root_dir,
translations_dir=translations_dir,
language_codes=language_codes,
excluded_dirs=excluded_dirs,
)
# Run cleanup
removed = manager.cleanup_orphaned_translations(markdown=False, images=True)
# Should have removed one file
assert removed == 1
# Valid translation should still exist
assert valid_trans_img.exists()
# Orphaned translation should be removed
assert not orphaned_trans_img.exists()
@@ -41,7 +41,6 @@ def project_translator(temp_project_dir):
"co_op_translator.config.llm_config.config.LLMConfig.get_available_provider"
) as mock_get_provider,
):
# Setup mock translators and config
mock_text_translator.create.return_value = MagicMock()
mock_markdown_translator.create.return_value = MagicMock()
@@ -49,113 +48,20 @@ def project_translator(temp_project_dir):
mock_get_provider.return_value = "azure" # Mock LLM provider
translator = ProjectTranslator("ko ja", root_dir=temp_project_dir)
# Mock async methods
translator.translate_all_markdown_files = AsyncMock(return_value=(2, []))
translator.translate_all_image_files = AsyncMock(return_value=(2, []))
return translator
@pytest.mark.asyncio
async def test_translate_markdown(project_translator, temp_project_dir):
"""Test translating a single markdown file."""
# Setup
md_file = temp_project_dir / "docs" / "test.md"
project_translator.markdown_translator.translate_markdown = AsyncMock(
return_value="# Test Document\nThis is a test in Korean."
)
# Execute
await project_translator.translate_markdown(md_file, "ko")
# Verify
translated_file = temp_project_dir / "translations" / "ko" / "docs" / "test.md"
assert translated_file.exists()
content = translated_file.read_text(encoding="utf-8")
assert "Test Document" in content
@pytest.mark.asyncio
async def test_translate_image(project_translator, temp_project_dir):
"""Test translating a single image file."""
# Setup
image_file = temp_project_dir / "images" / "test.png"
project_translator.image_translator.translate_image = MagicMock(
return_value=str(temp_project_dir / "translated_images" / "ko_test.png")
)
# Execute
await project_translator.translate_image(image_file, "ko")
# Verify
project_translator.image_translator.translate_image.assert_called_once()
@pytest.mark.asyncio
async def test_translate_all_markdown_files(project_translator, temp_project_dir):
"""Test translating all markdown files."""
# Setup
project_translator.markdown_translator.translate_markdown = AsyncMock(
return_value="# 테스트 문서\n이것은 테스트입니다."
)
# Execute
await project_translator.translate_all_markdown_files()
# Verify
for lang in ["ko", "ja"]:
translated_file = temp_project_dir / "translations" / lang / "docs" / "test.md"
assert translated_file.exists()
@pytest.mark.asyncio
async def test_translate_all_image_files(project_translator, temp_project_dir):
"""Test translating all image files."""
# Setup
(temp_project_dir / "images").mkdir(parents=True, exist_ok=True)
image_file = temp_project_dir / "images" / "test.png"
image_file.touch() # Create a dummy file for testing
async def mock_translate_image(path, lang, _):
return str(temp_project_dir / "translated_images" / f"{lang}_test.png")
project_translator.image_translator.translate_image = AsyncMock(
side_effect=mock_translate_image
)
# Execute
await asyncio.wait_for(project_translator.translate_all_image_files(), timeout=10)
# Verify
assert (
project_translator.image_translator.translate_image.call_count == 2
) # Called for "ko" and "ja"
project_translator.image_translator.translate_image.assert_any_call(
image_file, "ko", ANY
)
project_translator.image_translator.translate_image.assert_any_call(
image_file, "ja", ANY
)
@pytest.mark.asyncio
async def test_translate_project_async(project_translator):
"""Test translating entire project asynchronously."""
# Setup
project_translator.translate_all_markdown_files = AsyncMock()
project_translator.translate_all_image_files = AsyncMock()
# Execute
await project_translator.translate_project_async(images=True, markdown=True)
# Verify
project_translator.translate_all_markdown_files.assert_called_once()
project_translator.translate_all_image_files.assert_called_once()
def test_translate_project(project_translator):
"""Test the synchronous translate_project method."""
# Setup
with patch.object(asyncio, "run") as mock_run:
with patch.object(asyncio, "run", side_effect=lambda x: None) as mock_run:
# Execute
project_translator.translate_project(images=True, markdown=True)
# Verify
mock_run.assert_called_once()
@@ -183,37 +89,6 @@ async def test_check_and_retry_translations(project_translator, temp_project_dir
assert translated_file.exists()
@pytest.mark.asyncio
async def test_process_api_requests_parallel(project_translator):
"""Test processing API requests in parallel."""
# Setup
mock_task1 = AsyncMock()
mock_task2 = AsyncMock()
tasks = [mock_task1, mock_task2]
# Execute
await project_translator.process_api_requests_parallel(tasks, "Test tasks")
# No direct verification possible due to queue-based implementation
# Success is indicated by no exceptions being raised
@pytest.mark.asyncio
async def test_process_api_requests_sequential(project_translator):
"""Test processing API requests sequentially."""
# Setup
mock_task1 = AsyncMock()
mock_task2 = AsyncMock()
tasks = [lambda: mock_task1(), lambda: mock_task2()]
# Execute
await project_translator.process_api_requests_sequential(tasks, "Test tasks")
# Verify
mock_task1.assert_called_once()
mock_task2.assert_called_once()
def test_markdown_only_mode(temp_project_dir):
"""Test ProjectTranslator in markdown-only mode."""
with (
@@ -227,7 +102,6 @@ def test_markdown_only_mode(temp_project_dir):
"co_op_translator.config.llm_config.config.LLMConfig.get_available_provider"
) as mock_get_provider,
):
# Setup mocks
mock_text_translator.create.return_value = MagicMock()
mock_markdown_translator.create.return_value = MagicMock()
@@ -0,0 +1,259 @@
import pytest
from unittest.mock import AsyncMock, MagicMock
from co_op_translator.core.project.translation_manager import TranslationManager
@pytest.fixture
def temp_project_dir(tmp_path):
"""Creates a temporary project directory structure."""
docs_dir = tmp_path / "docs"
docs_dir.mkdir()
(docs_dir / "test.md").write_text(
"# Test Document\nThis is a test.", encoding="utf-8"
)
images_dir = tmp_path / "images"
images_dir.mkdir()
(images_dir / "test.png").touch()
translations_dir = tmp_path / "translations"
translations_dir.mkdir()
return tmp_path
@pytest.fixture
def mock_translation_manager(temp_project_dir):
"""Creates a fully mocked instance of TranslationManager."""
manager = MagicMock(spec=TranslationManager)
# Mock all async methods
manager.translate_markdown = AsyncMock()
manager.translate_image = AsyncMock()
manager.translate_all_markdown_files = AsyncMock()
manager.translate_all_image_files = AsyncMock()
manager.process_api_requests_parallel = AsyncMock()
manager.process_api_requests_sequential = AsyncMock()
# Set up common attributes
manager.root_dir = temp_project_dir
manager.translations_dir = temp_project_dir / "translations"
manager.image_dir = temp_project_dir / "translated_images"
manager.language_codes = ["ko", "ja"]
return manager
@pytest.mark.asyncio
async def test_translate_markdown(mock_translation_manager, temp_project_dir):
"""Tests the translation of a single markdown file."""
md_file = temp_project_dir / "docs" / "test.md"
translated_content = "# Test Document\nThis is a translated test."
# Setup mock
mock_translation_manager.markdown_translator = MagicMock()
mock_translation_manager.markdown_translator.translate_markdown = AsyncMock(
return_value=translated_content
)
mock_translation_manager.translate_markdown = AsyncMock(
return_value=translated_content
)
# Call and verify
mock_translation_manager.translate_markdown.assert_not_called()
result = await mock_translation_manager.translate_markdown(md_file, "ko")
assert result == translated_content
mock_translation_manager.translate_markdown.assert_awaited_once_with(md_file, "ko")
@pytest.mark.asyncio
async def test_translate_image(mock_translation_manager, temp_project_dir):
"""Tests the translation of a single image file."""
image_file = temp_project_dir / "images" / "test.png"
expected_translated_path = str(
temp_project_dir / "translated_images" / "ko" / "images" / "test.png"
)
# Setup mock
mock_translation_manager.image_translator = MagicMock()
mock_translation_manager.image_translator.translate_image = MagicMock(
return_value=expected_translated_path
)
mock_translation_manager.translate_image = AsyncMock(
return_value=expected_translated_path
)
# Call and verify
mock_translation_manager.translate_image.assert_not_called()
result = await mock_translation_manager.translate_image(image_file, "ko")
assert result == expected_translated_path
mock_translation_manager.translate_image.assert_awaited_once_with(image_file, "ko")
@pytest.mark.asyncio
async def test_translate_all_markdown_files(mock_translation_manager, temp_project_dir):
"""Tests the translation of all markdown files."""
expected_count = 2 # One for each language
mock_translation_manager.translate_all_markdown_files = AsyncMock(
return_value=(expected_count, [])
)
# Call and verify
mock_translation_manager.translate_all_markdown_files.assert_not_called()
count, errors = await mock_translation_manager.translate_all_markdown_files()
assert count == expected_count
assert not errors
mock_translation_manager.translate_all_markdown_files.assert_awaited_once()
@pytest.mark.asyncio
async def test_translate_all_image_files(mock_translation_manager, temp_project_dir):
"""Tests the translation of all image files."""
expected_count = 2 # One for each language
mock_translation_manager.translate_all_image_files = AsyncMock(
return_value=(expected_count, [])
)
# Call and verify
mock_translation_manager.translate_all_image_files.assert_not_called()
count, errors = await mock_translation_manager.translate_all_image_files()
assert count == expected_count
assert not errors
mock_translation_manager.translate_all_image_files.assert_awaited_once()
@pytest.mark.asyncio
async def test_process_api_requests_parallel(mock_translation_manager):
"""Tests parallel API request processing."""
mock_tasks = [AsyncMock() for _ in range(3)]
mock_translation_manager.process_api_requests_parallel = AsyncMock()
await mock_translation_manager.process_api_requests_parallel(
mock_tasks, "Test tasks"
)
mock_translation_manager.process_api_requests_parallel.assert_called_once_with(
mock_tasks, "Test tasks"
)
@pytest.mark.asyncio
async def test_process_api_requests_sequential(mock_translation_manager):
"""Tests sequential API request processing."""
mock_tasks = [AsyncMock() for _ in range(3)]
mock_translation_manager.process_api_requests_sequential = AsyncMock()
await mock_translation_manager.process_api_requests_sequential(
mock_tasks, "Test tasks"
)
mock_translation_manager.process_api_requests_sequential.assert_called_once_with(
mock_tasks, "Test tasks"
)
@pytest.mark.asyncio
async def test_get_outdated_translations(mock_translation_manager, temp_project_dir):
"""Tests the detection of outdated translation files."""
# Setup test files
ko_dir = temp_project_dir / "translations" / "ko"
ko_dir.mkdir(parents=True, exist_ok=True)
test_md = temp_project_dir / "test.md"
test_md.write_text("# Test Document\nThis is a test.", encoding="utf-8")
ko_test_md = ko_dir / "test.md"
ko_test_md.parent.mkdir(parents=True, exist_ok=True)
ko_test_md.write_text("# 테스트 문서\n이것은 테스트입니다.", encoding="utf-8")
# Mock _is_translation_outdated to return True for our test file
mock_translation_manager._is_translation_outdated = MagicMock(return_value=True)
mock_translation_manager.get_outdated_translations = (
TranslationManager.get_outdated_translations.__get__(mock_translation_manager)
)
mock_translation_manager.root_dir = temp_project_dir
mock_translation_manager.translations_dir = temp_project_dir / "translations"
mock_translation_manager.language_codes = ["ko"]
# Get outdated translations
outdated_files = mock_translation_manager.get_outdated_translations()
# Verify results
assert len(outdated_files) == 1
original_file, translation_file = outdated_files[0]
assert original_file.name == "test.md"
assert translation_file.parent.name == "ko"
@pytest.mark.asyncio
async def test_retranslate_outdated_files(mock_translation_manager, temp_project_dir):
"""Tests retranslation of outdated files."""
# Setup test files
test_md = temp_project_dir / "test.md"
test_md.write_text("# Test Document\nThis is a test.", encoding="utf-8")
ko_dir = temp_project_dir / "translations" / "ko"
ko_dir.mkdir(parents=True, exist_ok=True)
ko_test_md = ko_dir / "test.md"
ko_test_md.write_text("# 테스트 문서\n이것은 테스트입니다.", encoding="utf-8")
# Create a list of outdated files
outdated_files = [(test_md, ko_test_md)]
# Mock translate_markdown
mock_translation_manager.markdown_translator = MagicMock()
mock_translation_manager.markdown_translator.translate_markdown = AsyncMock(
return_value="# 테스트 문서\n이것은 테스트입니다."
)
mock_translation_manager.markdown_translator.create_metadata = MagicMock(
return_value={"file_hash": "test_hash"}
)
mock_translation_manager.translate_markdown = AsyncMock(
return_value=str(ko_test_md)
)
mock_translation_manager.retranslate_outdated_files = (
TranslationManager.retranslate_outdated_files.__get__(mock_translation_manager)
)
mock_translation_manager.translations_dir = ko_dir.parent
# Call retranslate_outdated_files
await mock_translation_manager.retranslate_outdated_files(outdated_files)
# Verify results
mock_translation_manager.translate_markdown.assert_awaited_once_with(test_md, "ko")
@pytest.mark.asyncio
async def test_translate_project_async_with_outdated(
mock_translation_manager, temp_project_dir
):
"""Tests the full project translation process with outdated files."""
# Setup test files
test_md = temp_project_dir / "test.md"
test_md.write_text("# Test Document\nThis is a test.", encoding="utf-8")
# Mock necessary methods
mock_translation_manager.get_outdated_translations = MagicMock(
return_value=[(test_md, temp_project_dir / "translations" / "ko" / "test.md")]
)
mock_translation_manager.retranslate_outdated_files = AsyncMock()
mock_translation_manager.translate_all_markdown_files = AsyncMock()
mock_translation_manager.translate_all_image_files = AsyncMock()
mock_translation_manager.directory_manager = MagicMock()
mock_translation_manager.directory_manager.sync_directory_structure = MagicMock(
return_value=(0, 0, [])
)
mock_translation_manager.directory_manager.cleanup_orphaned_translations = (
MagicMock(return_value=0)
)
mock_translation_manager.translate_project_async = (
TranslationManager.translate_project_async.__get__(mock_translation_manager)
)
# Call translate_project_async
await mock_translation_manager.translate_project_async(markdown=True)
# Verify the sequence of operations
mock_translation_manager.directory_manager.sync_directory_structure.assert_called_once()
mock_translation_manager.directory_manager.cleanup_orphaned_translations.assert_called_once()
mock_translation_manager.get_outdated_translations.assert_called_once()
mock_translation_manager.retranslate_outdated_files.assert_called_once()
mock_translation_manager.translate_all_markdown_files.assert_called_once()
@@ -2,8 +2,6 @@
Test cases for file utility functions.
"""
import os
import shutil
from pathlib import Path
import pytest
from co_op_translator.utils.common.file_utils import (
@@ -0,0 +1,83 @@
"""
Tests for metadata_utils.py
"""
import json
from freezegun import freeze_time
from co_op_translator.utils.common.metadata_utils import (
calculate_file_hash,
create_metadata,
format_metadata_comment,
)
def test_calculate_file_hash(tmp_path):
# Create a temporary file with known content
test_file = tmp_path / "test.txt"
test_content = "Hello, World!"
test_file.write_text(test_content)
# Calculate hash
result = calculate_file_hash(test_file)
# MD5 hash of "Hello, World!" is 65a8e27d8879283831b664bd8b7f0ad4
expected_hash = "65a8e27d8879283831b664bd8b7f0ad4"
assert result == expected_hash
@freeze_time("2025-01-26T14:30:00Z") # Using UTC time
def test_create_metadata(tmp_path):
# Create a test file
test_file = tmp_path / "test.txt"
test_content = "Test content"
test_file.write_text(test_content)
# Test with root_dir
root_dir = tmp_path
result = create_metadata(test_file, "ko", root_dir)
expected = {
"original_hash": calculate_file_hash(test_file),
"translation_date": "2025-01-26T14:30:00+00:00", # UTC time
"source_file": "test.txt",
"language_code": "ko",
}
assert result == expected
# Test without root_dir
result_no_root = create_metadata(test_file, "en")
expected_no_root = {
"original_hash": calculate_file_hash(test_file),
"translation_date": "2025-01-26T14:30:00+00:00", # UTC time
"source_file": str(test_file),
"language_code": "en",
}
assert result_no_root == expected_no_root
def test_format_metadata_comment():
test_metadata = {
"original_hash": "test_hash",
"translation_date": "2025-01-26T14:30:00+09:00",
"source_file": "test.txt",
"language_code": "ko",
}
result = format_metadata_comment(test_metadata)
# Verify the format
assert result.startswith("<!--\nCO_OP_TRANSLATOR_METADATA:\n")
assert result.endswith("\n-->\n")
# Extract and parse the JSON content
json_content = result.replace("<!--\nCO_OP_TRANSLATOR_METADATA:\n", "").replace(
"\n-->\n", ""
)
parsed_metadata = json.loads(json_content)
# Verify the metadata content
assert parsed_metadata == test_metadata
# Verify the indentation
assert " " in result # Should have 2-space indentation