Core: Standardize Translated Image Hash Prefixing & Update Directory Manager Behavior (#325)

This commit is contained in:
Minseok Song
2026-01-02 10:42:20 +09:00
committed by GitHub
parent 8c34a25e9c
commit 892600cde8
4 changed files with 62 additions and 64 deletions
@@ -25,19 +25,27 @@ class DirectoryManager:
translations_dir: Path,
language_codes: list[str],
excluded_dirs: list[str],
image_dir: Path | None = None,
):
"""Initialize directory manager with project configuration.
Args:
root_dir: Root directory containing original files
translations_dir: Directory for translated files
translations_dir: Directory for translated text files (markdown/notebooks)
language_codes: List of target language codes
excluded_dirs: List of directories to exclude
image_dir: Directory for translated images (flat tree, language code embedded in filename)
"""
self.root_dir = root_dir
self.translations_dir = translations_dir
self.language_codes = language_codes
self.excluded_dirs = excluded_dirs
# Default to root_dir / "translated_images" if not provided
self.image_dir = (
image_dir
if image_dir is not None
else (self.root_dir / "translated_images")
)
def sync_directory_structure(
self, markdown: bool = True, images: bool = True, notebooks: bool = True
@@ -315,13 +323,12 @@ class DirectoryManager:
# Handle image files
if images:
# Collect all image files in the original directory
original_images = {} # path_hash -> original_file_path
# Collect all candidate original images (compute path hash map)
original_images: dict[str, Path] = {}
try:
for original_img_file in self.root_dir.rglob("*"):
if not original_img_file.is_file():
continue
if original_img_file.suffix.lower() not in [
".png",
".jpg",
@@ -329,7 +336,6 @@ class DirectoryManager:
".gif",
]:
continue
try:
path_hash = get_unique_id(original_img_file, self.root_dir)
original_images[path_hash] = original_img_file
@@ -338,26 +344,21 @@ class DirectoryManager:
except Exception as e:
logger.warning(f"Error scanning for original images: {e}")
for lang_code in self.language_codes:
translation_dir = self.translations_dir / lang_code
if not translation_dir.exists():
logger.info(
f"Image translation directory does not exist: {translation_dir}"
)
continue
image_dir = self.image_dir
if not image_dir.exists():
logger.info(f"Image directory does not exist: {image_dir}")
else:
logger.info(f"Checking translated images in: {image_dir}")
logger.info(f"Checking translated images in: {translation_dir}")
image_files = []
try:
image_files = list(translation_dir.rglob("*"))
image_files = list(image_dir.rglob("*"))
except Exception as e:
logger.warning(f"Error scanning for image files: {e}")
image_files = []
for image_file in image_files:
if not image_file.is_file():
continue
if image_file.suffix.lower() not in [
".png",
".jpg",
@@ -368,12 +369,10 @@ class DirectoryManager:
try:
parts = image_file.name.split(".")
if (
len(parts) < 4
): # Need at least name, hash/prefix, lang_code, and extension
# Expect at least: base, hash/prefix, lang, ext
if len(parts) < 4:
continue
# Last part is extension, second to last is lang_code, third to last is path hash or prefix
extension = parts[-1]
lang_code = parts[-2]
path_hash_segment = parts[-3]
@@ -381,7 +380,7 @@ class DirectoryManager:
if lang_code not in self.language_codes:
continue
# Determine if this translated image corresponds to any known original
# Match by full hash or by prefix
if len(path_hash_segment) < 64:
has_match = any(
h.startswith(path_hash_segment)
@@ -391,12 +390,18 @@ class DirectoryManager:
has_match = path_hash_segment in original_images
if not has_match:
image_file.unlink()
removed_count += 1
logger.debug(f"Removed orphaned image: {image_file}")
try:
image_file.unlink()
removed_count += 1
logger.debug(f"Removed orphaned image: {image_file}")
except Exception as e:
logger.warning(
f"Failed to delete orphaned image {image_file}: {e}"
)
continue
parent = image_file.parent
while parent != translation_dir:
while parent != image_dir:
if parent.exists() and not any(parent.iterdir()):
try:
parent.rmdir()
@@ -130,6 +130,7 @@ class ProjectTranslator:
self.translations_dir,
self.language_codes,
self.excluded_dirs,
image_dir=self.image_dir,
)
self.translation_manager = TranslationManager(
self.root_dir,
@@ -84,7 +84,11 @@ class TranslationManager:
self.translation_types = translation_types
self.add_disclaimer = add_disclaimer
self.directory_manager = DirectoryManager(
root_dir, translations_dir, language_codes, excluded_dirs
root_dir,
translations_dir,
language_codes,
excluded_dirs,
image_dir=image_dir,
)
async def translate_image(
+25 -37
View File
@@ -137,34 +137,11 @@ import logging
logger = logging.getLogger(__name__)
# Minimum and fallback hash prefix lengths for translated image filenames.
_HASH_PREFIX_LENGTHS = (16, 20, 24)
# Registry used to detect hash prefix collisions within a single process.
# Keys are (language_code, hash_prefix); values are the corresponding full hash.
_HASH_PREFIX_REGISTRY: dict[tuple[str, str], str] = {}
# Fixed hash prefix length (in hex characters) for translated image filenames.
HASH_PREFIX_LENGTH = 16
def _select_hash_prefix(full_hash: str, language_code: str) -> str:
"""Select a collision-aware hash prefix for a translated filename.
Uses the shared in-memory registry to prefer the shortest prefix length
while ensuring that (language_code, prefix) is never associated with two
different full hashes within a single process.
"""
hash_prefix = full_hash
for prefix_len in _HASH_PREFIX_LENGTHS:
candidate_prefix = full_hash[:prefix_len]
key = (language_code, candidate_prefix)
existing_full = _HASH_PREFIX_REGISTRY.get(key)
if existing_full is None or existing_full == full_hash:
_HASH_PREFIX_REGISTRY[key] = full_hash
hash_prefix = candidate_prefix
break
return hash_prefix
# Using a fixed-length hash prefix; collision-aware selection logic removed.
def read_input_file(input_file: str | Path) -> str:
@@ -350,8 +327,8 @@ def generate_translated_filename(
# Compute the full path hash based on the normalized path
full_hash = get_unique_id(str(original_filepath), root_dir)
# Choose the shortest available prefix that does not collide (per language)
hash_prefix = _select_hash_prefix(full_hash, language_code)
# Use a fixed-size prefix for deterministic filenames across runs/OS
hash_prefix = full_hash[:HASH_PREFIX_LENGTH]
# Generate the new filename with the selected hash prefix and language code
new_filename = f"{original_filename}.{hash_prefix}.{language_code}{file_ext}"
@@ -412,13 +389,18 @@ def filter_files(directory: str | Path, excluded_dirs, extension: str = None) ->
def migrate_translated_image_filenames(
image_dir: Path, language_codes: list[str]
) -> dict[str, str]:
"""Rename translated images that still use full 64-hex hashes in filenames.
"""Rename translated images to use a fixed 16-hex hash prefix.
This helper only operates within the given image_dir and for the specified
language codes. Files already using truncated hashes are left untouched.
This helper operates within the given image_dir and for the specified
language codes.
It handles the following cases:
- Legacy filenames that embed a full 64-hex path hash → shorten to first 16 hex
- Older truncated prefixes with 20 or 24 hex → shorten to first 16 hex
Files already using a 16-hex prefix are left untouched.
Returns a mapping from old basenames to new basenames for use when
updating markdown links.
updating markdown/notebook links.
"""
image_dir = Path(image_dir)
@@ -455,14 +437,20 @@ def migrate_translated_image_filenames(
if lang_code not in language_codes:
continue
# Only migrate legacy filenames that embed a full 64-hex path hash.
if not re.fullmatch(r"[0-9a-f]{64}", raw_hash):
# Operate on hex segments of length 64 (legacy full) or 24/20 (older truncation).
if not re.fullmatch(r"[0-9a-f]+", raw_hash):
continue
seg_len = len(raw_hash)
if seg_len == HASH_PREFIX_LENGTH:
# Already standardized to 16 hex; nothing to do
continue
if seg_len not in (64, 24, 20):
# Unknown pattern length; skip to be conservative
continue
base_name = ".".join(parts[:-3])
full_hash = raw_hash
hash_prefix = _select_hash_prefix(full_hash, lang_code)
new_name = f"{base_name}.{hash_prefix}.{lang_code}.{extension}"
new_prefix = raw_hash[:HASH_PREFIX_LENGTH]
new_name = f"{base_name}.{new_prefix}.{lang_code}.{extension}"
if new_name == image_file.name:
continue