mirror of
https://github.com/Azure/co-op-translator
synced 2026-08-09 12:00:08 +00:00
Core: Preserve markdown chunk formatting for oversized lists (#397)
This commit is contained in:
@@ -406,26 +406,11 @@ def split_markdown_content(content: str, max_tokens: int, tokenizer) -> list:
|
||||
if "@@CODE_BLOCK" in line or "@@INLINE_CODE" in line:
|
||||
chunks.append(line)
|
||||
else:
|
||||
words = line.split()
|
||||
word_chunk = []
|
||||
word_chunk_tokens = 0
|
||||
|
||||
for word in words:
|
||||
word_with_space = word + " "
|
||||
word_tokens = count_tokens(
|
||||
word_with_space, tokenizer
|
||||
chunks.extend(
|
||||
_split_text_preserving_whitespace(
|
||||
line, max_tokens, tokenizer
|
||||
)
|
||||
|
||||
if word_chunk_tokens + word_tokens <= max_tokens:
|
||||
word_chunk.append(word_with_space)
|
||||
word_chunk_tokens += word_tokens
|
||||
else:
|
||||
chunks.append("".join(word_chunk))
|
||||
word_chunk = [word_with_space]
|
||||
word_chunk_tokens = word_tokens
|
||||
|
||||
if word_chunk:
|
||||
chunks.append("".join(word_chunk))
|
||||
)
|
||||
else:
|
||||
current_chunk = [line]
|
||||
current_length = line_tokens
|
||||
@@ -443,20 +428,82 @@ def split_markdown_content(content: str, max_tokens: int, tokenizer) -> list:
|
||||
return chunks
|
||||
|
||||
|
||||
def _split_text_preserving_whitespace(
|
||||
text: str, max_tokens: int, tokenizer
|
||||
) -> list[str]:
|
||||
"""Split oversized text without normalizing whitespace or line breaks."""
|
||||
chunks: list[str] = []
|
||||
current_parts: list[str] = []
|
||||
current_tokens = 0
|
||||
|
||||
for part in re.findall(r"\s+|\S+", text):
|
||||
part_tokens = count_tokens(part, tokenizer)
|
||||
|
||||
if part_tokens > max_tokens:
|
||||
if current_parts:
|
||||
chunks.append("".join(current_parts))
|
||||
current_parts = []
|
||||
current_tokens = 0
|
||||
|
||||
chunks.extend(
|
||||
_split_unbroken_text_preserving_characters(part, max_tokens, tokenizer)
|
||||
)
|
||||
continue
|
||||
|
||||
if current_tokens + part_tokens <= max_tokens:
|
||||
current_parts.append(part)
|
||||
current_tokens += part_tokens
|
||||
else:
|
||||
chunks.append("".join(current_parts))
|
||||
current_parts = [part]
|
||||
current_tokens = part_tokens
|
||||
|
||||
if current_parts:
|
||||
chunks.append("".join(current_parts))
|
||||
|
||||
return chunks
|
||||
|
||||
|
||||
def _split_unbroken_text_preserving_characters(
|
||||
text: str, max_tokens: int, tokenizer
|
||||
) -> list[str]:
|
||||
"""Split a single oversized non-whitespace span while preserving all chars."""
|
||||
chunks: list[str] = []
|
||||
current_chars: list[str] = []
|
||||
current_tokens = 0
|
||||
|
||||
for char in text:
|
||||
char_tokens = count_tokens(char, tokenizer)
|
||||
|
||||
if current_chars and current_tokens + char_tokens > max_tokens:
|
||||
chunks.append("".join(current_chars))
|
||||
current_chars = [char]
|
||||
current_tokens = char_tokens
|
||||
else:
|
||||
current_chars.append(char)
|
||||
current_tokens += char_tokens
|
||||
|
||||
if current_chars:
|
||||
chunks.append("".join(current_chars))
|
||||
|
||||
return chunks
|
||||
|
||||
|
||||
def _group_lines_preserving_list_items(text: str) -> list[str]:
|
||||
"""Group markdown text into split units while keeping list-item blocks together."""
|
||||
"""Group markdown text while keeping each list item's continuation together."""
|
||||
lines = text.splitlines(keepends=True)
|
||||
if not lines:
|
||||
return []
|
||||
|
||||
grouped_lines: list[str] = []
|
||||
idx = 0
|
||||
list_item_pattern = re.compile(r"^\s{0,3}(?:[*+-]|\d+[.)])\s+")
|
||||
list_item_pattern = re.compile(r"^(\s*)(?:[*+-]|\d+[.)])\s+")
|
||||
|
||||
while idx < len(lines):
|
||||
line = lines[idx]
|
||||
list_item_match = list_item_pattern.match(line)
|
||||
|
||||
if list_item_pattern.match(line):
|
||||
if list_item_match:
|
||||
block = [line]
|
||||
idx += 1
|
||||
|
||||
@@ -467,9 +514,11 @@ def _group_lines_preserving_list_items(text: str) -> list[str]:
|
||||
idx += 1
|
||||
continue
|
||||
|
||||
if next_line.startswith((" ", "\t")) or list_item_pattern.match(
|
||||
next_line
|
||||
):
|
||||
next_item_match = list_item_pattern.match(next_line)
|
||||
if next_item_match:
|
||||
break
|
||||
|
||||
if next_line.startswith((" ", "\t")):
|
||||
block.append(next_line)
|
||||
idx += 1
|
||||
continue
|
||||
@@ -691,9 +740,7 @@ def build_translated_image_link(
|
||||
new_filename = generate_translated_filename(
|
||||
actual_image_path, language_code, root_dir
|
||||
)
|
||||
return os.path.join(rel_path, language_code, new_filename).replace(
|
||||
os.path.sep, "/"
|
||||
)
|
||||
return os.path.join(rel_path, language_code, new_filename).replace(os.path.sep, "/")
|
||||
|
||||
|
||||
def _slugify_heading_text(text: str) -> str:
|
||||
|
||||
@@ -12,6 +12,7 @@ from co_op_translator.utils.llm.markdown_utils import (
|
||||
generate_prompt_template,
|
||||
count_links_in_markdown,
|
||||
split_markdown_content,
|
||||
_group_lines_preserving_list_items,
|
||||
update_notebook_links,
|
||||
normalize_cjk_emphasis_markers,
|
||||
normalize_internal_anchor_links,
|
||||
@@ -174,6 +175,141 @@ def test_process_markdown_with_many_links():
|
||||
assert all(count_links_in_markdown(chunk) <= max_links for chunk in chunks)
|
||||
|
||||
|
||||
def test_group_lines_preserving_list_items_splits_sibling_items():
|
||||
"""Sibling list items should be separate split units for large TOCs."""
|
||||
content = """- [One](one.md)
|
||||
- [Two](two.md)
|
||||
- [Three](three.md)
|
||||
|
||||
Paragraph
|
||||
"""
|
||||
|
||||
groups = _group_lines_preserving_list_items(content)
|
||||
|
||||
assert groups == [
|
||||
"- [One](one.md)\n",
|
||||
"- [Two](two.md)\n",
|
||||
"- [Three](three.md)\n\n",
|
||||
"Paragraph\n",
|
||||
]
|
||||
|
||||
|
||||
def test_group_lines_preserving_list_items_keeps_nested_code_with_item():
|
||||
"""Indented code blocks inside a list item should stay with that item."""
|
||||
content = """- Step 1
|
||||
|
||||
```bash
|
||||
echo one
|
||||
```
|
||||
- Step 2
|
||||
"""
|
||||
|
||||
groups = _group_lines_preserving_list_items(content)
|
||||
|
||||
assert groups == [
|
||||
"- Step 1\n\n ```bash\n echo one\n ```\n",
|
||||
"- Step 2\n",
|
||||
]
|
||||
|
||||
|
||||
def test_group_lines_preserving_list_items_splits_nested_items():
|
||||
"""Nested list items should still be separate split units."""
|
||||
content = """- Parent
|
||||
- Child 1
|
||||
- Grandchild
|
||||
- Next
|
||||
"""
|
||||
|
||||
groups = _group_lines_preserving_list_items(content)
|
||||
|
||||
assert groups == [
|
||||
"- Parent\n",
|
||||
" - Child 1\n",
|
||||
" - Grandchild\n",
|
||||
"- Next\n",
|
||||
]
|
||||
|
||||
|
||||
def test_split_markdown_content_keeps_list_newlines_when_splitting_large_toc():
|
||||
"""Oversized link-heavy lists should split at item boundaries, not words."""
|
||||
content = """- [One](one.md)
|
||||
- [Two](two.md)
|
||||
- [Three](three.md)
|
||||
- [Four](four.md)
|
||||
"""
|
||||
|
||||
class MockTokenizer:
|
||||
def encode(self, text):
|
||||
return list(text)
|
||||
|
||||
chunks = split_markdown_content(content, 35, MockTokenizer())
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert all(" - " not in chunk for chunk in chunks)
|
||||
assert "".join(chunks) == content
|
||||
|
||||
|
||||
def test_split_markdown_content_keeps_nested_toc_newlines_when_splitting():
|
||||
"""Large nested TOCs should not be flattened into one-line chunks."""
|
||||
content = """- Phi application development samples
|
||||
- Text & Chat Applications
|
||||
- Phi-4 Samples
|
||||
- [Chat With Phi-4-mini ONNX Model](./md/02.Application/01.TextAndChat/Phi4/ChatWithPhi4ONNX/README.md)
|
||||
- [Chat with Phi-4 local ONNX Model .NET](./md/04.HOL/dotnet/src/LabsPhi4-Chat-01OnnxRuntime/)
|
||||
- Azure AI Inference SDK Code Based Samples
|
||||
- Phi-4 Samples
|
||||
- [Generate project code using Phi-4-multimodal](./md/02.Application/02.Code/Phi4/GenProjectCode/README.md)
|
||||
"""
|
||||
|
||||
class MockTokenizer:
|
||||
def encode(self, text):
|
||||
return list(text)
|
||||
|
||||
chunks = split_markdown_content(content, 120, MockTokenizer())
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert (
|
||||
"- Phi application development samples - Text & Chat Applications"
|
||||
not in "".join(chunks)
|
||||
)
|
||||
assert "- Phi-4 Samples - [Chat With Phi-4-mini ONNX Model]" not in "".join(chunks)
|
||||
assert "".join(chunks) == content
|
||||
|
||||
|
||||
def test_split_markdown_content_preserves_whitespace_in_oversized_fallback():
|
||||
"""Fallback splitting should preserve exact whitespace and newlines."""
|
||||
content = """- Step with a very long continuation
|
||||
This continuation keeps double spaces.
|
||||
This continuation keeps the second line.
|
||||
"""
|
||||
|
||||
class MockTokenizer:
|
||||
def encode(self, text):
|
||||
return list(text)
|
||||
|
||||
chunks = split_markdown_content(content, 32, MockTokenizer())
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert "".join(chunks) == content
|
||||
assert " double spaces" in "".join(chunks)
|
||||
assert "\n This continuation keeps the second line." in "".join(chunks)
|
||||
|
||||
|
||||
def test_split_markdown_content_splits_unbroken_text_without_losing_characters():
|
||||
"""Fallback splitting should preserve oversized spans without whitespace."""
|
||||
content = "https://example.com/" + ("verylongpath" * 8) + "\n"
|
||||
|
||||
class MockTokenizer:
|
||||
def encode(self, text):
|
||||
return list(text)
|
||||
|
||||
chunks = split_markdown_content(content, 25, MockTokenizer())
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert "".join(chunks) == content
|
||||
assert all(len(chunk) <= 25 for chunk in chunks if chunk.strip())
|
||||
|
||||
|
||||
def test_generate_prompt_template():
|
||||
"""Test generating translation prompt template."""
|
||||
document_chunk = "Test content"
|
||||
|
||||
Reference in New Issue
Block a user