Skip to content

Instantly share code, notes, and snippets.

@havardgulldahl
Last active September 3, 2026 01:09
Show Gist options
  • Select an option

  • Save havardgulldahl/2a7ef3c440d2f0d934c4139259a21cfa to your computer and use it in GitHub Desktop.

Select an option

Save havardgulldahl/2a7ef3c440d2f0d934c4139259a21cfa to your computer and use it in GitHub Desktop.
A python script to translate a pdf inline, where the translated strings are overlayed on top of the original text -- like translate.google.com will do. See usage instructions at the bottom.
[project]
name = "pdf-argos-translator"
version = "0.1.1"
requires-python = ">=3.10"
dependencies = [
"pymupdf>=1.23.0",
"argostranslate>=1.9.0",
"beautifulsoup4>=4.12.0",
"lxml>=5.0.0",
]
[project.scripts]
pdf-translate = "translate_pdf:main"
html-translate = "translate_html:main"
[tool.setuptools]
py-modules = ["translate_pdf", "translate_html", "translate_common"]
"""
Shared utilities for the translate_pdf and translate_html scripts.
Covers:
- Application path / models-directory helpers
- Argos model discovery, installation, and download
- Language code listing
"""
import os
import sys
import shutil
import argostranslate.translate # noqa: F401 – ensure argostranslate is importable
# Common ISO 639-1 language names for --language-codes output
LANGUAGE_NAMES = {
"ar": "Arabic",
"az": "Azerbaijani",
"bg": "Bulgarian",
"bn": "Bengali",
"ca": "Catalan",
"cs": "Czech",
"da": "Danish",
"de": "German",
"el": "Greek",
"en": "English",
"eo": "Esperanto",
"es": "Spanish",
"et": "Estonian",
"fa": "Persian",
"fi": "Finnish",
"fr": "French",
"ga": "Irish",
"he": "Hebrew",
"hi": "Hindi",
"hu": "Hungarian",
"id": "Indonesian",
"it": "Italian",
"ja": "Japanese",
"ko": "Korean",
"lt": "Lithuanian",
"lv": "Latvian",
"nl": "Dutch",
"no": "Norwegian",
"pl": "Polish",
"pt": "Portuguese",
"ro": "Romanian",
"ru": "Russian",
"sk": "Slovak",
"sl": "Slovenian",
"sv": "Swedish",
"tr": "Turkish",
"uk": "Ukrainian",
"vi": "Vietnamese",
"zh": "Chinese",
}
def get_application_path():
"""Works for normal script and PyInstaller executable."""
if getattr(sys, "frozen", False):
return os.path.dirname(sys.executable)
return os.path.dirname(os.path.abspath(__file__))
def get_models_dir() -> str:
"""Return the per-user cache directory for downloaded Argos models."""
configured_dir = os.environ.get("PDF_ARGOS_CACHE_DIR")
if configured_dir:
models_dir = configured_dir
elif sys.platform == "win32":
cache_root = os.environ.get(
"LOCALAPPDATA", os.path.expanduser(r"~\AppData\Local")
)
models_dir = os.path.join(cache_root, "pdf-argos-translator", "models")
elif sys.platform == "darwin":
models_dir = os.path.expanduser("~/Library/Caches/pdf-argos-translator/models")
else:
cache_root = os.environ.get("XDG_CACHE_HOME", os.path.expanduser("~/.cache"))
models_dir = os.path.join(cache_root, "pdf-argos-translator", "models")
os.makedirs(models_dir, exist_ok=True)
return models_dir
def build_model_filename(from_code, to_code):
"""
Preferred model naming scheme:
model.<from>-<to>.argosmodel
Example: model.et-en.argosmodel
"""
return f"model.{from_code}-{to_code}.argosmodel"
def is_pair_installed(from_code, to_code):
import argostranslate.package
installed = argostranslate.package.get_installed_packages()
return any(p.from_code == from_code and p.to_code == to_code for p in installed)
def find_local_model_file(from_code, to_code):
"""
Search for a local model file using preferred + legacy names.
"""
app_dir = get_application_path()
models_dir = get_models_dir()
preferred = build_model_filename(from_code, to_code)
candidates = [
# Preferred scheme
os.path.join(models_dir, preferred),
os.path.join(app_dir, preferred),
# Common Argos naming
os.path.join(models_dir, f"translate-{from_code}_{to_code}.argosmodel"),
os.path.join(app_dir, f"translate-{from_code}_{to_code}.argosmodel"),
os.path.join(models_dir, f"{from_code}_{to_code}.argosmodel"),
os.path.join(app_dir, f"{from_code}_{to_code}.argosmodel"),
# Legacy names
os.path.join(app_dir, "model.argos"),
os.path.join(app_dir, "model.argosmodel"),
os.path.join(models_dir, "model.argos"),
os.path.join(models_dir, "model.argosmodel"),
]
for path in candidates:
if os.path.exists(path):
return path
return None
def install_local_or_download_model(from_code, to_code):
"""
Ensure requested Argos language pair is installed:
1) use already-installed pair
2) try local model file
3) auto-download from package index
"""
import argostranslate.package
if is_pair_installed(from_code, to_code):
print(f"Language model already installed: {from_code}->{to_code}")
return
local_model = find_local_model_file(from_code, to_code)
if local_model:
print(f"Installing local model: {local_model}")
argostranslate.package.install_from_path(local_model)
if is_pair_installed(from_code, to_code):
print(f"Installed local model for {from_code}->{to_code}")
return
print("Local model did not match requested pair; trying auto-download...")
print(f"Auto-downloading model for {from_code}->{to_code} ...")
try:
argostranslate.package.update_package_index()
available = argostranslate.package.get_available_packages()
except Exception as e:
print(f"CRITICAL ERROR: Cannot update Argos package index: {e}")
sys.exit(1)
pkg = next(
(p for p in available if p.from_code == from_code and p.to_code == to_code),
None,
)
if pkg is None:
print(f"CRITICAL ERROR: No Argos model available for {from_code}->{to_code}.")
print("Run with --language-codes to inspect available pairs.")
sys.exit(1)
try:
downloaded_path = pkg.download()
argostranslate.package.install_from_path(downloaded_path)
print(f"Downloaded and installed {from_code}->{to_code}")
# Cache under preferred naming scheme for easier offline reuse
cached_path = os.path.join(
get_models_dir(), build_model_filename(from_code, to_code)
)
try:
shutil.copy2(downloaded_path, cached_path)
print(f"Cached model as: {cached_path}")
except Exception as cache_err:
print(f"Warning: could not cache model file: {cache_err}")
except Exception as e:
print(f"CRITICAL ERROR: Failed to download/install model: {e}")
sys.exit(1)
if not is_pair_installed(from_code, to_code):
print(f"CRITICAL ERROR: Model still not installed for {from_code}->{to_code}.")
sys.exit(1)
def list_language_codes():
"""
Print:
- Common language abbreviations
- Installed Argos pairs
- Downloadable Argos pairs (online)
"""
import argostranslate.package
print("=== Common language codes (ISO 639-1) ===")
for code in sorted(LANGUAGE_NAMES):
print(f"{code:>3} - {LANGUAGE_NAMES[code]}")
print()
print("=== Installed Argos translation pairs ===")
try:
installed = argostranslate.package.get_installed_packages()
if not installed:
print(" (none installed)")
else:
for p in sorted(installed, key=lambda x: (x.from_code, x.to_code)):
print(f" {p.from_code} -> {p.to_code}")
except Exception as e:
print(f" Could not read installed packages: {e}")
print()
print("=== Downloadable Argos translation pairs (online index) ===")
try:
argostranslate.package.update_package_index()
available = argostranslate.package.get_available_packages()
if not available:
print(" (none returned)")
else:
pairs = sorted({(p.from_code, p.to_code) for p in available})
for src, dst in pairs:
print(f" {src} -> {dst}")
except Exception as e:
print(f" Could not load online index: {e}")
print(" (network unavailable or blocked)")
print()
"""
DOM-level HTML translator using Argos Translate.
Walks the HTML document tree and translates every visible text node in-place,
preserving all markup, attributes, scripts, and styles untouched.
Strategy
--------
Rather than translating each text node in isolation (which gives poor quality
for words inside inline tags like <b> or <i> that lack sentence context), the
translator works at the *block* level:
1. Every "leaf block" element (a block-level tag containing no nested block
tags, e.g. <p>, <h1>, <li>, <td> …) is translated as a single unit.
2. Inline child elements (<b>, <i>, <a>, <em>, <strong>, <span> …) are
temporarily replaced by opaque placeholder tokens ``ZT{n}ZT`` so the
surrounding sentence is translated with full context.
3. The inline elements' own text is then translated independently (with
terminal punctuation stripped so it fits naturally back into the sentence).
4. Placeholder tokens are substituted back with the (now translated) inline
elements to reconstruct the markup.
5. Any text node not covered by a leaf-block pass (rare orphan nodes) is
translated individually as a fallback.
Quote normalisation
-------------------
Argos sometimes collapses adjacent double-quote characters. A post-processing
step collapses any run of 2+ consecutive ASCII double-quotes to a single one.
Usage:
python translate_html.py input.html [options]
Examples:
python translate_html.py page.html --from ru --to en
python translate_html.py page.html --from ru --to en --output translated.html
python translate_html.py --language-codes
"""
import argparse
import os
import re
import sys
import argostranslate.translate
from bs4 import BeautifulSoup, NavigableString, Comment, Tag
from translate_common import (
install_local_or_download_model,
list_language_codes,
)
# ---------------------------------------------------------------------------
# Tag classification sets
# ---------------------------------------------------------------------------
# Tags whose text content must not be translated
_SKIP_TAGS = frozenset(
[
"script",
"style",
"code",
"pre",
"kbd",
"samp",
"var",
"math",
"svg",
"template",
"noscript",
]
)
# Block-level tags that represent a complete translatable unit (a "sentence
# container"). Leaf instances of these are translated as one string.
_BLOCK_TAGS = frozenset(
[
"address",
"article",
"aside",
"blockquote",
"caption",
"dd",
"details",
"dialog",
"dt",
"figcaption",
"figure",
"footer",
"h1",
"h2",
"h3",
"h4",
"h5",
"h6",
"header",
"label",
"legend",
"li",
"main",
"p",
"section",
"summary",
"td",
"th",
"button",
# <div> is included so that divs used as paragraphs are covered; nested
# divs are naturally excluded by the "leaf block" rule.
"div",
]
)
# Inline tags whose content belongs to the surrounding sentence. They are
# replaced with placeholders during block translation and then reinserted.
_INLINE_TAGS = frozenset(
[
"a",
"abbr",
"acronym",
"b",
"bdo",
"big",
"cite",
"data",
"del",
"dfn",
"em",
"i",
"ins",
"mark",
"q",
"s",
"small",
"span",
"strong",
"sub",
"sup",
"time",
"tt",
"u",
"wbr",
]
)
# Regex matching our placeholder tokens
_PLACEHOLDER_RE = re.compile(r"ZT(\d+)ZT")
# Terminal punctuation that standalone inline translations may spuriously add
_TERMINAL_PUNCT_RE = re.compile(r"[.!?]+$")
# ---------------------------------------------------------------------------
# Low-level translation helpers
# ---------------------------------------------------------------------------
def _raw_translate(text, from_code, to_code):
"""Thin wrapper around argostranslate with quote normalisation."""
result = argostranslate.translate.translate(text, from_code, to_code)
# Collapse runs of 2+ consecutive ASCII double-quotes to a single one.
result = re.sub(r'"{2,}', '"', result)
return result
def _translate_inline_text(text, from_code, to_code):
"""
Translate a short inline snippet, stripping any terminal punctuation that
the model may add when given an isolated word or phrase.
"""
translated = _raw_translate(text, from_code, to_code)
# Strip trailing sentence-ending punctuation that doesn't belong inside
# an inline element embedded in a larger sentence.
translated = _TERMINAL_PUNCT_RE.sub("", translated).rstrip()
return translated
# ---------------------------------------------------------------------------
# Block-level translation logic
# ---------------------------------------------------------------------------
def _is_leaf_block(element):
"""Return True if *element* contains no block-level descendants."""
for desc in element.descendants:
if isinstance(desc, Tag) and desc.name in _BLOCK_TAGS:
return False
return True
def _has_translatable_text(element):
"""Return True if *element* contains at least one non-whitespace text node
that is not inside a skip tag."""
for node in element.descendants:
if not isinstance(node, NavigableString) or isinstance(node, Comment):
continue
if not str(node).strip():
continue
# Check ancestry for skip tags
skip = False
for parent in node.parents:
if parent is element:
break
if isinstance(parent, Tag) and parent.name in _SKIP_TAGS:
skip = True
break
if not skip:
return True
return False
def _translate_inline_element(element, from_code, to_code):
"""
Recursively translate the content of an inline element.
If the element contains only text (no further inline children), translate
the text directly. If it contains nested inline elements, apply the
placeholder approach recursively so nested formatting is preserved.
"""
# Collect direct children
children = list(element.children)
# Check if there are any inline sub-elements
has_nested_inline = any(
isinstance(c, Tag) and c.name in _INLINE_TAGS for c in children
)
if not has_nested_inline:
# Simple case: translate full text content
inner = element.get_text()
if inner.strip():
translated = _translate_inline_text(inner.strip(), from_code, to_code)
element.clear()
element.append(NavigableString(translated))
return
# Check if this element has any *direct* text content (i.e. text not
# coming from a nested inline). When the element is a pure wrapper like
# <b><i>text</i></b> there is no direct text, so building a placeholder
# string of just "ZT0ZT" would be meaningless to translate. In that case,
# fall back to translating the full text content and replacing all children.
has_direct_text = any(
isinstance(c, NavigableString) and not isinstance(c, Comment) and str(c).strip()
for c in children
)
if not has_direct_text:
inner = element.get_text()
if inner.strip():
translated = _translate_inline_text(inner.strip(), from_code, to_code)
element.clear()
element.append(NavigableString(translated))
return
# Recursive case: use placeholder approach on the inline element itself
parts = []
inline_map = {}
for child in children:
if isinstance(child, NavigableString):
if not isinstance(child, Comment):
parts.append(str(child))
elif isinstance(child, Tag):
if child.name in _SKIP_TAGS:
parts.append(str(child))
elif child.name in _INLINE_TAGS:
n = len(inline_map)
inline_map[n] = child
parts.append(f"ZT{n}ZT")
else:
parts.append(element.get_text()) # fallback
raw = "".join(parts).strip()
if not raw:
return
translated_block = _translate_inline_text(raw, from_code, to_code)
for n, child_elem in inline_map.items():
_translate_inline_element(child_elem, from_code, to_code)
# Reconstruct element
element.clear()
tokens = _PLACEHOLDER_RE.split(translated_block)
for i, tok in enumerate(tokens):
if i % 2 == 0:
if tok:
element.append(NavigableString(tok))
else:
n = int(tok)
if n in inline_map:
element.append(inline_map[n])
def _translate_leaf_block(element, from_code, to_code):
"""
Translate a leaf block element using the placeholder strategy.
Direct text children are included verbatim; direct inline-tag children are
replaced with ZT{n}ZT tokens. The resulting string is translated as one
unit (full sentence context). Each inline element is then translated
independently (with terminal punctuation stripped), and the placeholders
are substituted back to reconstruct the element.
If the NMT drops a placeholder (rare), that inline element is silently
omitted from the output rather than producing corrupt HTML.
"""
parts = []
inline_map = {} # placeholder index -> Tag
for child in list(element.children):
if isinstance(child, Comment):
continue
if isinstance(child, NavigableString):
parts.append(str(child))
elif isinstance(child, Tag):
if child.name in _SKIP_TAGS:
# Keep skip-tag content literally in the placeholder string so
# it doesn't get sent to the NMT; extract and reattach later.
n = len(inline_map)
inline_map[n] = child
parts.append(f"ZT{n}ZT")
elif child.name in _INLINE_TAGS:
n = len(inline_map)
inline_map[n] = child
parts.append(f"ZT{n}ZT")
else:
# Nested block — shouldn't appear in a leaf block, but if it
# does, include its text inline.
parts.append(child.get_text())
raw = "".join(parts).strip()
if not raw:
return
# ------------------------------------------------------------------
# If there are no inline elements, a single translation suffices.
# ------------------------------------------------------------------
if not inline_map:
translated = _raw_translate(raw, from_code, to_code)
element["title"] = raw
element.clear()
element.append(NavigableString(translated))
return
# ------------------------------------------------------------------
# Translate block with placeholders (preserves surrounding context).
# ------------------------------------------------------------------
translated = _raw_translate(raw, from_code, to_code)
# ------------------------------------------------------------------
# Sanity: check that placeholders were not significantly displaced by
# the NMT. If the relative position of any placeholder in the output
# differs from its source position by more than 40% of the text length,
# the template translation is unreliable for that placeholder's context
# extraction (the model rearranged the sentence around the placeholder).
# We record which placeholders are "stable" for context extraction.
stable_placeholders = (
set()
) # int indices of placeholders not significantly displaced
for n in inline_map:
ph = f"ZT{n}ZT"
src_pos = raw.find(ph)
out_pos = translated.find(ph)
if src_pos == -1 or out_pos == -1:
continue
src_ratio = src_pos / max(len(raw), 1)
out_ratio = out_pos / max(len(translated), 1)
if abs(src_ratio - out_ratio) <= 0.40:
stable_placeholders.add(n)
# ------------------------------------------------------------------
# Also translate the block with inline content embedded (no placeholders)
# so we can extract context-aware translations for inline elements.
# The user explicitly permits multiple translation passes.
# ------------------------------------------------------------------
full_parts = []
for child in list(element.children):
if isinstance(child, Comment):
continue
if isinstance(child, NavigableString):
full_parts.append(str(child))
elif isinstance(child, Tag):
if child.name not in _SKIP_TAGS:
full_parts.append(child.get_text())
raw_full = "".join(full_parts).strip()
# Only bother with the second translation when the texts differ
translated_full = (
_raw_translate(raw_full, from_code, to_code) if raw_full != raw else None
)
# ------------------------------------------------------------------
# For each inline element: try context-aware extraction first, then
# fall back to translating the element independently.
# ------------------------------------------------------------------
for n, child_elem in inline_map.items():
if child_elem.name in _SKIP_TAGS:
continue # leave skip-tag content untouched
context_text = None
if translated_full and n in stable_placeholders:
# Pass the source inline text's word count as a lower bound so
# we don't accept a single-word fragment (like "The") when the
# NMT rearranged the sentence and the placeholder moved to the front.
src_words = max(1, len(child_elem.get_text().split()))
context_text = _match_context_span(
translated, translated_full, n, min_words=src_words
)
if context_text is not None:
# Context extraction succeeded: overwrite the element's content
# with the contextually-correct translation. Any nested inline
# formatting is lost here, but context accuracy takes priority.
child_elem.clear()
child_elem.append(NavigableString(context_text))
else:
# Extraction failed: translate the inline element's content
# independently (works well for short words / phrases).
_translate_inline_element(child_elem, from_code, to_code)
# Reconstruct the element from the translated string + inline elements.
# Store the original plain text as a title attribute so users can hover
# to see the source text.
original_text = raw_full if raw_full else raw
element["title"] = original_text
element.clear()
tokens = _PLACEHOLDER_RE.split(translated)
for i, tok in enumerate(tokens):
if i % 2 == 0:
if tok:
element.append(NavigableString(tok))
else:
n = int(tok)
if n in inline_map:
element.append(inline_map[n])
# else: placeholder was dropped by NMT – silently skip
def _match_context_span(template_trans, full_trans, placeholder_n, min_words=1):
"""
Given a translated template string with ``ZT{n}ZT`` placeholders and the
translation of the same block *without* any placeholders, try to extract
the span in *full_trans* that corresponds to placeholder *placeholder_n*.
Uses ``\\b`` word-boundary anchors (so "agrees" does not match inside
"disagrees"). Both left and right boundaries must be located for the
result to be reliable.
*min_words*: minimum word count for the extracted span. Pass the source
inline element's word count to avoid returning a 1-word fragment like
"The" when the NMT moved the placeholder to the sentence start.
Returns the extracted span (stripped), or ``None`` if extraction fails.
"""
tokens = _PLACEHOLDER_RE.split(template_trans)
ph_pos = None
for i in range(1, len(tokens), 2):
if int(tokens[i]) == placeholder_n:
ph_pos = i
break
if ph_pos is None:
return None
left_ctx = tokens[ph_pos - 1] if ph_pos > 0 else ""
right_ctx = tokens[ph_pos + 1] if ph_pos + 1 < len(tokens) else ""
full_lower = full_trans.lower()
def _seq_end(words, haystack, start=0):
# Strip terminal punctuation so "headline." doesn't break \b matching
clean = [re.sub(r"[.,!?;:]+$", "", w) for w in words]
clean = [w for w in clean if w]
if not clean:
return -1
anchor = r"\s+".join(re.escape(w) for w in clean)
m = re.search(r"\b" + anchor + r"\b", haystack[start:], re.IGNORECASE)
return start + m.end() if m else -1
def _seq_start(words, haystack, start=0):
# Strip terminal punctuation so "headline." doesn't break \b matching
clean = [re.sub(r"[.,!?;:]+$", "", w) for w in words]
clean = [w for w in clean if w]
if not clean:
return -1
anchor = r"\s+".join(re.escape(w) for w in clean)
m = re.search(r"\b" + anchor + r"\b", haystack[start:], re.IGNORECASE)
return start + m.start() if m else -1
# ----- Left boundary ------------------------------------------------
left_end = None
if not left_ctx.strip():
left_end = 0 # placeholder is at the template start
else:
left_words = left_ctx.strip().split()
for n_words in range(min(4, len(left_words)), 0, -1):
pos = _seq_end(left_words[-n_words:], full_lower)
if pos != -1:
while pos < len(full_trans) and full_trans[pos] == " ":
pos += 1
left_end = pos
break
# ----- Right boundary -----------------------------------------------
right_start = None
if not right_ctx.strip():
right_start = len(full_trans) # placeholder is at the template end
else:
right_words = right_ctx.strip().split()
search_from = left_end if left_end is not None else 0
for n_words in range(min(4, len(right_words)), 0, -1):
pos = _seq_start(right_words[:n_words], full_lower, search_from)
if pos != -1:
while pos > search_from and full_trans[pos - 1] == " ":
pos -= 1
right_start = pos
break
if left_end is None or right_start is None:
return None
if left_end >= right_start:
return None
# When the placeholder was at the template start (left_end=0) and the
# right boundary lands in the first 20% of the full translation, the NMT
# likely rearranged the sentence — the span would be just an article or
# similar fragment, so reject.
if left_end == 0 and not left_ctx.strip():
if right_start < len(full_trans) * 0.20:
return None
span = full_trans[left_end:right_start].strip()
if not span:
return None
if len(span) > len(full_trans) * 0.80:
return None
if len(span.split()) < min_words:
return None
return span
def _should_translate_node(node):
"""
Return True when a NavigableString carries translatable human text
(used for the fallback per-node path).
"""
if isinstance(node, Comment):
return False
text = str(node)
if not text.strip():
return False
for parent in node.parents:
if hasattr(parent, "name") and parent.name in _SKIP_TAGS:
return False
return True
def translate_html_string(html_content, from_code, to_code):
"""
Parse *html_content*, translate every visible text node, and return the
modified HTML as a string.
Block-level elements are translated as complete units (preserving inline
markup via placeholders). Any remaining orphan text nodes are translated
individually as a fallback.
"""
soup = BeautifulSoup(html_content, "lxml")
# ------------------------------------------------------------------
# Pass 1: translate leaf block elements as whole units
# ------------------------------------------------------------------
# Collect leaf blocks first to avoid mutating the tree while iterating.
leaf_blocks = [
elem
for elem in soup.find_all(_BLOCK_TAGS)
if _is_leaf_block(elem) and _has_translatable_text(elem)
]
# Track which text nodes have already been translated so the fallback
# pass doesn't double-translate them.
translated_blocks = set()
total_blocks = len(leaf_blocks)
print(f"Found {total_blocks} translatable leaf-block elements.")
for idx, elem in enumerate(leaf_blocks, start=1):
try:
_translate_leaf_block(elem, from_code, to_code)
translated_blocks.add(id(elem))
except Exception as e:
print(f" [{idx}/{total_blocks}] Warning: block translation error: {e}")
if idx % 50 == 0 or idx == total_blocks:
print(f" Translated {idx}/{total_blocks} blocks...")
# ------------------------------------------------------------------
# Pass 2: fallback – translate any remaining orphan text nodes that
# were not covered by a leaf-block ancestor.
# ------------------------------------------------------------------
orphan_nodes = []
for node in soup.find_all(string=True):
if not _should_translate_node(node):
continue
# Check if any ancestor was a translated leaf block
covered = False
for parent in node.parents:
if id(parent) in translated_blocks:
covered = True
break
if not covered:
orphan_nodes.append(node)
if orphan_nodes:
print(f"Found {len(orphan_nodes)} orphan text nodes (fallback).")
for idx, node in enumerate(orphan_nodes, start=1):
original = str(node)
clean = original.strip()
if not clean:
continue
try:
translated = _raw_translate(clean, from_code, to_code)
except Exception as e:
print(f" Warning: skipping orphan node (translation error): {e}")
continue
leading = original[: len(original) - len(original.lstrip())]
trailing = original[len(original.rstrip()) :]
node.replace_with(NavigableString(leading + translated + trailing))
return str(soup)
def translate_html_file(input_path, output_path, from_code, to_code):
"""
Read *input_path*, translate its text nodes, and write to *output_path*.
"""
install_local_or_download_model(from_code, to_code)
print(f"Reading: {input_path}")
with open(input_path, "r", encoding="utf-8", errors="replace") as fh:
html_content = fh.read()
print(f"Translating {from_code} -> {to_code} ...")
translated_html = translate_html_string(html_content, from_code, to_code)
with open(output_path, "w", encoding="utf-8") as fh:
fh.write(translated_html)
print(f"Done! Saved to: {output_path}")
def main():
parser = argparse.ArgumentParser(
description="Translate an HTML file at the DOM level using Argos Translate.",
)
parser.add_argument(
"input_file",
nargs="?",
help="Path to the input HTML file.",
)
parser.add_argument(
"--from",
dest="from_lang",
default="ru",
help="Source language code (default: ru).",
)
parser.add_argument(
"--to",
dest="to_lang",
default="en",
help="Target language code (default: en).",
)
parser.add_argument(
"--output",
dest="output_file",
default=None,
help=(
"Path for the translated HTML output. "
"Defaults to <input_stem>_translated.html next to the input file."
),
)
parser.add_argument(
"--language-codes",
action="store_true",
help="List language abbreviations and Argos translation pairs, then exit.",
)
args = parser.parse_args()
if args.language_codes:
list_language_codes()
return
if not args.input_file:
parser.error(
"the following arguments are required: input_file "
"(unless --language-codes is used)"
)
input_file = args.input_file
if args.output_file:
output_file = args.output_file
else:
stem, _ = os.path.splitext(input_file)
output_file = stem + "_translated.html"
translate_html_file(
input_path=input_file,
output_path=output_file,
from_code=args.from_lang,
to_code=args.to_lang,
)
if __name__ == "__main__":
main()
import os
import sys
import shutil
import argparse
import subprocess
import tempfile
import fitz # PyMuPDF
import argostranslate.translate
from translate_common import (
install_local_or_download_model,
list_language_codes,
)
def find_ghostscript_executable():
"""
Find Ghostscript executable across platforms.
"""
candidates = ["gs", "gswin64c", "gswin32c"]
for c in candidates:
path = shutil.which(c)
if path:
return path
return None
def run_ghostscript_compress(input_pdf, output_pdf, preset="ebook"):
"""
Compress PDF with Ghostscript.
Returns True on success, False on failure.
"""
gs_exe = find_ghostscript_executable()
if not gs_exe:
print("Warning: Ghostscript not found (gs). Skipping GS compression.")
return False
cmd = [
gs_exe,
"-sDEVICE=pdfwrite",
"-dCompatibilityLevel=1.6",
"-dNOPAUSE",
"-dQUIET",
"-dBATCH",
"-dDetectDuplicateImages=true",
"-dCompressFonts=true",
]
if preset != "default":
cmd.append(f"-dPDFSETTINGS=/{preset}")
cmd.append(f"-sOutputFile={output_pdf}")
cmd.append(input_pdf)
print(f"Running Ghostscript compression (preset: {preset}) ...")
try:
subprocess.run(cmd, check=True)
return True
except subprocess.CalledProcessError as e:
print(f"Warning: Ghostscript compression failed: {e}")
return False
def translate_and_overlay_blocks(
input_path,
output_path,
from_code,
to_code,
keep_original,
use_ghostscript=True,
gs_preset="ebook",
):
install_local_or_download_model(from_code, to_code)
doc = fitz.open(input_path)
print(f"Processing {len(doc)} pages...")
for page_num, page in enumerate(doc):
print(f"Translating page {page_num + 1}/{len(doc)}...")
blocks = page.get_text("blocks")
translated_blocks = [] # list of (rect, translated_text)
# 1) Collect translations and (optionally) redaction annotations
for block in blocks:
if len(block) < 7:
continue
x0, y0, x1, y1, text, _block_no, block_type = block[:7]
if block_type != 0:
continue
if not text or not text.strip():
continue
clean_text = text.replace("\n", " ").strip()
if len(clean_text) <= 1:
continue
try:
translated_text = argostranslate.translate.translate(
clean_text, from_code, to_code
)
except Exception as e:
print(f" Skipping block (translation error): {e}")
continue
rect = fitz.Rect(x0, y0, x1, y1)
translated_blocks.append((rect, translated_text))
if not keep_original:
page.add_redact_annot(rect, fill=(1, 1, 1))
# 2) Apply redactions ONCE per page
if not keep_original:
try:
# Keep images unchanged to reduce bloat from redaction processing
try:
page.apply_redactions(images=0)
except TypeError:
page.apply_redactions()
except Exception as e:
print(f" Warning: apply_redactions failed on page {page_num + 1}: {e}")
# 3) Insert translated text
for rect, translated_text in translated_blocks:
font_size = 11.0
res = -1.0
while res < 0 and font_size > 5:
res = page.insert_textbox(
rect,
translated_text,
fontsize=font_size,
fontname="helv",
color=(0, 0, 0),
align=0,
)
if res < 0:
font_size -= 0.5
# Save intermediate translated PDF (optimized PyMuPDF save)
tmp_fd, tmp_pdf_path = tempfile.mkstemp(suffix=".pdf")
os.close(tmp_fd)
save_kwargs = dict(
garbage=4,
deflate=True,
clean=True,
incremental=False,
)
try:
try:
doc.save(
tmp_pdf_path,
**save_kwargs,
deflate_images=True,
deflate_fonts=True,
)
except TypeError:
doc.save(tmp_pdf_path, **save_kwargs)
finally:
doc.close()
# Optional Ghostscript final compression
if use_ghostscript:
ok = run_ghostscript_compress(tmp_pdf_path, output_path, preset=gs_preset)
if not ok:
# Fallback: keep PyMuPDF output
shutil.move(tmp_pdf_path, output_path)
print("Saved without Ghostscript compression due to GS issue.")
else:
try:
os.remove(tmp_pdf_path)
except OSError:
pass
else:
shutil.move(tmp_pdf_path, output_path)
print(f"Done! Saved to: {output_path}")
def main():
parser = argparse.ArgumentParser()
parser.add_argument(
"input_file",
nargs="?",
help="Path to the input PDF file.",
)
parser.add_argument(
"--from",
dest="from_lang",
default="ru",
help="Source language code (default: ru).",
)
parser.add_argument(
"--to",
dest="to_lang",
default="en",
help="Target language code (default: en).",
)
parser.add_argument(
"--keep-original",
action="store_true",
help="Keep original text layer beneath translation (no redaction).",
)
parser.add_argument(
"--language-codes",
action="store_true",
help="List language abbreviations and Argos translation pairs, then exit.",
)
parser.add_argument(
"--no-gs",
action="store_true",
help="Disable Ghostscript compression pass.",
)
parser.add_argument(
"--gs-preset",
choices=["screen", "ebook", "printer", "prepress", "default"],
default="ebook",
help="Ghostscript PDFSETTINGS preset (default: ebook).",
)
args = parser.parse_args()
if args.language_codes:
list_language_codes()
return
if not args.input_file:
parser.error(
"the following arguments are required: input_file (unless --language-codes is used)"
)
input_file = args.input_file
output_file = os.path.splitext(input_file)[0] + "_translated.pdf"
translate_and_overlay_blocks(
input_path=input_file,
output_path=output_file,
from_code=args.from_lang,
to_code=args.to_lang,
keep_original=args.keep_original,
use_ghostscript=not args.no_gs,
gs_preset=args.gs_preset,
)
if __name__ == "__main__":
main()
@havardgulldahl

havardgulldahl commented Sep 3, 2026 •

Copy link
Copy Markdown
Author

Updated script, and made it installable by uv tool install:

uv tool install git+https://gist.github.com/havardgulldahl/2a7ef3c440d2f0d934c4139259a21cfa.git # (this gist)
pdf-translate input.pdf --from ru --to en --no-gs
html-translate input.html --from ru --to en

Get uv from here: https://docs.astral.sh/uv/getting-started/installation/

If you have Ghostscript on your machine, drop the --no-gs for considerable size reductions.

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment