mirror of
https://github.com/Azure/co-op-translator
synced 2026-08-09 12:00:08 +00:00
Core: Standardize Translated Image Hash Prefixing & Update Directory Manager Behavior (#325)
This commit is contained in:
@@ -25,19 +25,27 @@ class DirectoryManager:
|
||||
translations_dir: Path,
|
||||
language_codes: list[str],
|
||||
excluded_dirs: list[str],
|
||||
image_dir: Path | None = None,
|
||||
):
|
||||
"""Initialize directory manager with project configuration.
|
||||
|
||||
Args:
|
||||
root_dir: Root directory containing original files
|
||||
translations_dir: Directory for translated files
|
||||
translations_dir: Directory for translated text files (markdown/notebooks)
|
||||
language_codes: List of target language codes
|
||||
excluded_dirs: List of directories to exclude
|
||||
image_dir: Directory for translated images (flat tree, language code embedded in filename)
|
||||
"""
|
||||
self.root_dir = root_dir
|
||||
self.translations_dir = translations_dir
|
||||
self.language_codes = language_codes
|
||||
self.excluded_dirs = excluded_dirs
|
||||
# Default to root_dir / "translated_images" if not provided
|
||||
self.image_dir = (
|
||||
image_dir
|
||||
if image_dir is not None
|
||||
else (self.root_dir / "translated_images")
|
||||
)
|
||||
|
||||
def sync_directory_structure(
|
||||
self, markdown: bool = True, images: bool = True, notebooks: bool = True
|
||||
@@ -315,13 +323,12 @@ class DirectoryManager:
|
||||
|
||||
# Handle image files
|
||||
if images:
|
||||
# Collect all image files in the original directory
|
||||
original_images = {} # path_hash -> original_file_path
|
||||
# Collect all candidate original images (compute path hash map)
|
||||
original_images: dict[str, Path] = {}
|
||||
try:
|
||||
for original_img_file in self.root_dir.rglob("*"):
|
||||
if not original_img_file.is_file():
|
||||
continue
|
||||
|
||||
if original_img_file.suffix.lower() not in [
|
||||
".png",
|
||||
".jpg",
|
||||
@@ -329,7 +336,6 @@ class DirectoryManager:
|
||||
".gif",
|
||||
]:
|
||||
continue
|
||||
|
||||
try:
|
||||
path_hash = get_unique_id(original_img_file, self.root_dir)
|
||||
original_images[path_hash] = original_img_file
|
||||
@@ -338,26 +344,21 @@ class DirectoryManager:
|
||||
except Exception as e:
|
||||
logger.warning(f"Error scanning for original images: {e}")
|
||||
|
||||
for lang_code in self.language_codes:
|
||||
translation_dir = self.translations_dir / lang_code
|
||||
if not translation_dir.exists():
|
||||
logger.info(
|
||||
f"Image translation directory does not exist: {translation_dir}"
|
||||
)
|
||||
continue
|
||||
image_dir = self.image_dir
|
||||
if not image_dir.exists():
|
||||
logger.info(f"Image directory does not exist: {image_dir}")
|
||||
else:
|
||||
logger.info(f"Checking translated images in: {image_dir}")
|
||||
|
||||
logger.info(f"Checking translated images in: {translation_dir}")
|
||||
|
||||
image_files = []
|
||||
try:
|
||||
image_files = list(translation_dir.rglob("*"))
|
||||
image_files = list(image_dir.rglob("*"))
|
||||
except Exception as e:
|
||||
logger.warning(f"Error scanning for image files: {e}")
|
||||
image_files = []
|
||||
|
||||
for image_file in image_files:
|
||||
if not image_file.is_file():
|
||||
continue
|
||||
|
||||
if image_file.suffix.lower() not in [
|
||||
".png",
|
||||
".jpg",
|
||||
@@ -368,12 +369,10 @@ class DirectoryManager:
|
||||
|
||||
try:
|
||||
parts = image_file.name.split(".")
|
||||
if (
|
||||
len(parts) < 4
|
||||
): # Need at least name, hash/prefix, lang_code, and extension
|
||||
# Expect at least: base, hash/prefix, lang, ext
|
||||
if len(parts) < 4:
|
||||
continue
|
||||
|
||||
# Last part is extension, second to last is lang_code, third to last is path hash or prefix
|
||||
extension = parts[-1]
|
||||
lang_code = parts[-2]
|
||||
path_hash_segment = parts[-3]
|
||||
@@ -381,7 +380,7 @@ class DirectoryManager:
|
||||
if lang_code not in self.language_codes:
|
||||
continue
|
||||
|
||||
# Determine if this translated image corresponds to any known original
|
||||
# Match by full hash or by prefix
|
||||
if len(path_hash_segment) < 64:
|
||||
has_match = any(
|
||||
h.startswith(path_hash_segment)
|
||||
@@ -391,12 +390,18 @@ class DirectoryManager:
|
||||
has_match = path_hash_segment in original_images
|
||||
|
||||
if not has_match:
|
||||
image_file.unlink()
|
||||
removed_count += 1
|
||||
logger.debug(f"Removed orphaned image: {image_file}")
|
||||
try:
|
||||
image_file.unlink()
|
||||
removed_count += 1
|
||||
logger.debug(f"Removed orphaned image: {image_file}")
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
f"Failed to delete orphaned image {image_file}: {e}"
|
||||
)
|
||||
continue
|
||||
|
||||
parent = image_file.parent
|
||||
while parent != translation_dir:
|
||||
while parent != image_dir:
|
||||
if parent.exists() and not any(parent.iterdir()):
|
||||
try:
|
||||
parent.rmdir()
|
||||
|
||||
@@ -130,6 +130,7 @@ class ProjectTranslator:
|
||||
self.translations_dir,
|
||||
self.language_codes,
|
||||
self.excluded_dirs,
|
||||
image_dir=self.image_dir,
|
||||
)
|
||||
self.translation_manager = TranslationManager(
|
||||
self.root_dir,
|
||||
|
||||
@@ -84,7 +84,11 @@ class TranslationManager:
|
||||
self.translation_types = translation_types
|
||||
self.add_disclaimer = add_disclaimer
|
||||
self.directory_manager = DirectoryManager(
|
||||
root_dir, translations_dir, language_codes, excluded_dirs
|
||||
root_dir,
|
||||
translations_dir,
|
||||
language_codes,
|
||||
excluded_dirs,
|
||||
image_dir=image_dir,
|
||||
)
|
||||
|
||||
async def translate_image(
|
||||
|
||||
@@ -137,34 +137,11 @@ import logging
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Minimum and fallback hash prefix lengths for translated image filenames.
|
||||
_HASH_PREFIX_LENGTHS = (16, 20, 24)
|
||||
|
||||
# Registry used to detect hash prefix collisions within a single process.
|
||||
# Keys are (language_code, hash_prefix); values are the corresponding full hash.
|
||||
_HASH_PREFIX_REGISTRY: dict[tuple[str, str], str] = {}
|
||||
# Fixed hash prefix length (in hex characters) for translated image filenames.
|
||||
HASH_PREFIX_LENGTH = 16
|
||||
|
||||
|
||||
def _select_hash_prefix(full_hash: str, language_code: str) -> str:
|
||||
"""Select a collision-aware hash prefix for a translated filename.
|
||||
|
||||
Uses the shared in-memory registry to prefer the shortest prefix length
|
||||
while ensuring that (language_code, prefix) is never associated with two
|
||||
different full hashes within a single process.
|
||||
"""
|
||||
|
||||
hash_prefix = full_hash
|
||||
for prefix_len in _HASH_PREFIX_LENGTHS:
|
||||
candidate_prefix = full_hash[:prefix_len]
|
||||
key = (language_code, candidate_prefix)
|
||||
existing_full = _HASH_PREFIX_REGISTRY.get(key)
|
||||
|
||||
if existing_full is None or existing_full == full_hash:
|
||||
_HASH_PREFIX_REGISTRY[key] = full_hash
|
||||
hash_prefix = candidate_prefix
|
||||
break
|
||||
|
||||
return hash_prefix
|
||||
# Using a fixed-length hash prefix; collision-aware selection logic removed.
|
||||
|
||||
|
||||
def read_input_file(input_file: str | Path) -> str:
|
||||
@@ -350,8 +327,8 @@ def generate_translated_filename(
|
||||
# Compute the full path hash based on the normalized path
|
||||
full_hash = get_unique_id(str(original_filepath), root_dir)
|
||||
|
||||
# Choose the shortest available prefix that does not collide (per language)
|
||||
hash_prefix = _select_hash_prefix(full_hash, language_code)
|
||||
# Use a fixed-size prefix for deterministic filenames across runs/OS
|
||||
hash_prefix = full_hash[:HASH_PREFIX_LENGTH]
|
||||
|
||||
# Generate the new filename with the selected hash prefix and language code
|
||||
new_filename = f"{original_filename}.{hash_prefix}.{language_code}{file_ext}"
|
||||
@@ -412,13 +389,18 @@ def filter_files(directory: str | Path, excluded_dirs, extension: str = None) ->
|
||||
def migrate_translated_image_filenames(
|
||||
image_dir: Path, language_codes: list[str]
|
||||
) -> dict[str, str]:
|
||||
"""Rename translated images that still use full 64-hex hashes in filenames.
|
||||
"""Rename translated images to use a fixed 16-hex hash prefix.
|
||||
|
||||
This helper only operates within the given image_dir and for the specified
|
||||
language codes. Files already using truncated hashes are left untouched.
|
||||
This helper operates within the given image_dir and for the specified
|
||||
language codes.
|
||||
|
||||
It handles the following cases:
|
||||
- Legacy filenames that embed a full 64-hex path hash → shorten to first 16 hex
|
||||
- Older truncated prefixes with 20 or 24 hex → shorten to first 16 hex
|
||||
Files already using a 16-hex prefix are left untouched.
|
||||
|
||||
Returns a mapping from old basenames to new basenames for use when
|
||||
updating markdown links.
|
||||
updating markdown/notebook links.
|
||||
"""
|
||||
|
||||
image_dir = Path(image_dir)
|
||||
@@ -455,14 +437,20 @@ def migrate_translated_image_filenames(
|
||||
if lang_code not in language_codes:
|
||||
continue
|
||||
|
||||
# Only migrate legacy filenames that embed a full 64-hex path hash.
|
||||
if not re.fullmatch(r"[0-9a-f]{64}", raw_hash):
|
||||
# Operate on hex segments of length 64 (legacy full) or 24/20 (older truncation).
|
||||
if not re.fullmatch(r"[0-9a-f]+", raw_hash):
|
||||
continue
|
||||
seg_len = len(raw_hash)
|
||||
if seg_len == HASH_PREFIX_LENGTH:
|
||||
# Already standardized to 16 hex; nothing to do
|
||||
continue
|
||||
if seg_len not in (64, 24, 20):
|
||||
# Unknown pattern length; skip to be conservative
|
||||
continue
|
||||
|
||||
base_name = ".".join(parts[:-3])
|
||||
full_hash = raw_hash
|
||||
hash_prefix = _select_hash_prefix(full_hash, lang_code)
|
||||
new_name = f"{base_name}.{hash_prefix}.{lang_code}.{extension}"
|
||||
new_prefix = raw_hash[:HASH_PREFIX_LENGTH]
|
||||
new_name = f"{base_name}.{new_prefix}.{lang_code}.{extension}"
|
||||
|
||||
if new_name == image_file.name:
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user