Last active
September 3, 2026 01:09
-
-
Save havardgulldahl/2a7ef3c440d2f0d934c4139259a21cfa to your computer and use it in GitHub Desktop.
A python script to translate a pdf inline, where the translated strings are overlayed on top of the original text -- like translate.google.com will do. See usage instructions at the bottom.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| [project] | |
| name = "pdf-argos-translator" | |
| version = "0.1.1" | |
| requires-python = ">=3.10" | |
| dependencies = [ | |
| "pymupdf>=1.23.0", | |
| "argostranslate>=1.9.0", | |
| "beautifulsoup4>=4.12.0", | |
| "lxml>=5.0.0", | |
| ] | |
| [project.scripts] | |
| pdf-translate = "translate_pdf:main" | |
| html-translate = "translate_html:main" | |
| [tool.setuptools] | |
| py-modules = ["translate_pdf", "translate_html", "translate_common"] |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| """ | |
| Shared utilities for the translate_pdf and translate_html scripts. | |
| Covers: | |
| - Application path / models-directory helpers | |
| - Argos model discovery, installation, and download | |
| - Language code listing | |
| """ | |
| import os | |
| import sys | |
| import shutil | |
| import argostranslate.translate # noqa: F401 – ensure argostranslate is importable | |
| # Common ISO 639-1 language names for --language-codes output | |
| LANGUAGE_NAMES = { | |
| "ar": "Arabic", | |
| "az": "Azerbaijani", | |
| "bg": "Bulgarian", | |
| "bn": "Bengali", | |
| "ca": "Catalan", | |
| "cs": "Czech", | |
| "da": "Danish", | |
| "de": "German", | |
| "el": "Greek", | |
| "en": "English", | |
| "eo": "Esperanto", | |
| "es": "Spanish", | |
| "et": "Estonian", | |
| "fa": "Persian", | |
| "fi": "Finnish", | |
| "fr": "French", | |
| "ga": "Irish", | |
| "he": "Hebrew", | |
| "hi": "Hindi", | |
| "hu": "Hungarian", | |
| "id": "Indonesian", | |
| "it": "Italian", | |
| "ja": "Japanese", | |
| "ko": "Korean", | |
| "lt": "Lithuanian", | |
| "lv": "Latvian", | |
| "nl": "Dutch", | |
| "no": "Norwegian", | |
| "pl": "Polish", | |
| "pt": "Portuguese", | |
| "ro": "Romanian", | |
| "ru": "Russian", | |
| "sk": "Slovak", | |
| "sl": "Slovenian", | |
| "sv": "Swedish", | |
| "tr": "Turkish", | |
| "uk": "Ukrainian", | |
| "vi": "Vietnamese", | |
| "zh": "Chinese", | |
| } | |
| def get_application_path(): | |
| """Works for normal script and PyInstaller executable.""" | |
| if getattr(sys, "frozen", False): | |
| return os.path.dirname(sys.executable) | |
| return os.path.dirname(os.path.abspath(__file__)) | |
| def get_models_dir() -> str: | |
| """Return the per-user cache directory for downloaded Argos models.""" | |
| configured_dir = os.environ.get("PDF_ARGOS_CACHE_DIR") | |
| if configured_dir: | |
| models_dir = configured_dir | |
| elif sys.platform == "win32": | |
| cache_root = os.environ.get( | |
| "LOCALAPPDATA", os.path.expanduser(r"~\AppData\Local") | |
| ) | |
| models_dir = os.path.join(cache_root, "pdf-argos-translator", "models") | |
| elif sys.platform == "darwin": | |
| models_dir = os.path.expanduser("~/Library/Caches/pdf-argos-translator/models") | |
| else: | |
| cache_root = os.environ.get("XDG_CACHE_HOME", os.path.expanduser("~/.cache")) | |
| models_dir = os.path.join(cache_root, "pdf-argos-translator", "models") | |
| os.makedirs(models_dir, exist_ok=True) | |
| return models_dir | |
| def build_model_filename(from_code, to_code): | |
| """ | |
| Preferred model naming scheme: | |
| model.<from>-<to>.argosmodel | |
| Example: model.et-en.argosmodel | |
| """ | |
| return f"model.{from_code}-{to_code}.argosmodel" | |
| def is_pair_installed(from_code, to_code): | |
| import argostranslate.package | |
| installed = argostranslate.package.get_installed_packages() | |
| return any(p.from_code == from_code and p.to_code == to_code for p in installed) | |
| def find_local_model_file(from_code, to_code): | |
| """ | |
| Search for a local model file using preferred + legacy names. | |
| """ | |
| app_dir = get_application_path() | |
| models_dir = get_models_dir() | |
| preferred = build_model_filename(from_code, to_code) | |
| candidates = [ | |
| # Preferred scheme | |
| os.path.join(models_dir, preferred), | |
| os.path.join(app_dir, preferred), | |
| # Common Argos naming | |
| os.path.join(models_dir, f"translate-{from_code}_{to_code}.argosmodel"), | |
| os.path.join(app_dir, f"translate-{from_code}_{to_code}.argosmodel"), | |
| os.path.join(models_dir, f"{from_code}_{to_code}.argosmodel"), | |
| os.path.join(app_dir, f"{from_code}_{to_code}.argosmodel"), | |
| # Legacy names | |
| os.path.join(app_dir, "model.argos"), | |
| os.path.join(app_dir, "model.argosmodel"), | |
| os.path.join(models_dir, "model.argos"), | |
| os.path.join(models_dir, "model.argosmodel"), | |
| ] | |
| for path in candidates: | |
| if os.path.exists(path): | |
| return path | |
| return None | |
| def install_local_or_download_model(from_code, to_code): | |
| """ | |
| Ensure requested Argos language pair is installed: | |
| 1) use already-installed pair | |
| 2) try local model file | |
| 3) auto-download from package index | |
| """ | |
| import argostranslate.package | |
| if is_pair_installed(from_code, to_code): | |
| print(f"Language model already installed: {from_code}->{to_code}") | |
| return | |
| local_model = find_local_model_file(from_code, to_code) | |
| if local_model: | |
| print(f"Installing local model: {local_model}") | |
| argostranslate.package.install_from_path(local_model) | |
| if is_pair_installed(from_code, to_code): | |
| print(f"Installed local model for {from_code}->{to_code}") | |
| return | |
| print("Local model did not match requested pair; trying auto-download...") | |
| print(f"Auto-downloading model for {from_code}->{to_code} ...") | |
| try: | |
| argostranslate.package.update_package_index() | |
| available = argostranslate.package.get_available_packages() | |
| except Exception as e: | |
| print(f"CRITICAL ERROR: Cannot update Argos package index: {e}") | |
| sys.exit(1) | |
| pkg = next( | |
| (p for p in available if p.from_code == from_code and p.to_code == to_code), | |
| None, | |
| ) | |
| if pkg is None: | |
| print(f"CRITICAL ERROR: No Argos model available for {from_code}->{to_code}.") | |
| print("Run with --language-codes to inspect available pairs.") | |
| sys.exit(1) | |
| try: | |
| downloaded_path = pkg.download() | |
| argostranslate.package.install_from_path(downloaded_path) | |
| print(f"Downloaded and installed {from_code}->{to_code}") | |
| # Cache under preferred naming scheme for easier offline reuse | |
| cached_path = os.path.join( | |
| get_models_dir(), build_model_filename(from_code, to_code) | |
| ) | |
| try: | |
| shutil.copy2(downloaded_path, cached_path) | |
| print(f"Cached model as: {cached_path}") | |
| except Exception as cache_err: | |
| print(f"Warning: could not cache model file: {cache_err}") | |
| except Exception as e: | |
| print(f"CRITICAL ERROR: Failed to download/install model: {e}") | |
| sys.exit(1) | |
| if not is_pair_installed(from_code, to_code): | |
| print(f"CRITICAL ERROR: Model still not installed for {from_code}->{to_code}.") | |
| sys.exit(1) | |
| def list_language_codes(): | |
| """ | |
| Print: | |
| - Common language abbreviations | |
| - Installed Argos pairs | |
| - Downloadable Argos pairs (online) | |
| """ | |
| import argostranslate.package | |
| print("=== Common language codes (ISO 639-1) ===") | |
| for code in sorted(LANGUAGE_NAMES): | |
| print(f"{code:>3} - {LANGUAGE_NAMES[code]}") | |
| print() | |
| print("=== Installed Argos translation pairs ===") | |
| try: | |
| installed = argostranslate.package.get_installed_packages() | |
| if not installed: | |
| print(" (none installed)") | |
| else: | |
| for p in sorted(installed, key=lambda x: (x.from_code, x.to_code)): | |
| print(f" {p.from_code} -> {p.to_code}") | |
| except Exception as e: | |
| print(f" Could not read installed packages: {e}") | |
| print() | |
| print("=== Downloadable Argos translation pairs (online index) ===") | |
| try: | |
| argostranslate.package.update_package_index() | |
| available = argostranslate.package.get_available_packages() | |
| if not available: | |
| print(" (none returned)") | |
| else: | |
| pairs = sorted({(p.from_code, p.to_code) for p in available}) | |
| for src, dst in pairs: | |
| print(f" {src} -> {dst}") | |
| except Exception as e: | |
| print(f" Could not load online index: {e}") | |
| print(" (network unavailable or blocked)") | |
| print() |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| """ | |
| DOM-level HTML translator using Argos Translate. | |
| Walks the HTML document tree and translates every visible text node in-place, | |
| preserving all markup, attributes, scripts, and styles untouched. | |
| Strategy | |
| -------- | |
| Rather than translating each text node in isolation (which gives poor quality | |
| for words inside inline tags like <b> or <i> that lack sentence context), the | |
| translator works at the *block* level: | |
| 1. Every "leaf block" element (a block-level tag containing no nested block | |
| tags, e.g. <p>, <h1>, <li>, <td> …) is translated as a single unit. | |
| 2. Inline child elements (<b>, <i>, <a>, <em>, <strong>, <span> …) are | |
| temporarily replaced by opaque placeholder tokens ``ZT{n}ZT`` so the | |
| surrounding sentence is translated with full context. | |
| 3. The inline elements' own text is then translated independently (with | |
| terminal punctuation stripped so it fits naturally back into the sentence). | |
| 4. Placeholder tokens are substituted back with the (now translated) inline | |
| elements to reconstruct the markup. | |
| 5. Any text node not covered by a leaf-block pass (rare orphan nodes) is | |
| translated individually as a fallback. | |
| Quote normalisation | |
| ------------------- | |
| Argos sometimes collapses adjacent double-quote characters. A post-processing | |
| step collapses any run of 2+ consecutive ASCII double-quotes to a single one. | |
| Usage: | |
| python translate_html.py input.html [options] | |
| Examples: | |
| python translate_html.py page.html --from ru --to en | |
| python translate_html.py page.html --from ru --to en --output translated.html | |
| python translate_html.py --language-codes | |
| """ | |
| import argparse | |
| import os | |
| import re | |
| import sys | |
| import argostranslate.translate | |
| from bs4 import BeautifulSoup, NavigableString, Comment, Tag | |
| from translate_common import ( | |
| install_local_or_download_model, | |
| list_language_codes, | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # Tag classification sets | |
| # --------------------------------------------------------------------------- | |
| # Tags whose text content must not be translated | |
| _SKIP_TAGS = frozenset( | |
| [ | |
| "script", | |
| "style", | |
| "code", | |
| "pre", | |
| "kbd", | |
| "samp", | |
| "var", | |
| "math", | |
| "svg", | |
| "template", | |
| "noscript", | |
| ] | |
| ) | |
| # Block-level tags that represent a complete translatable unit (a "sentence | |
| # container"). Leaf instances of these are translated as one string. | |
| _BLOCK_TAGS = frozenset( | |
| [ | |
| "address", | |
| "article", | |
| "aside", | |
| "blockquote", | |
| "caption", | |
| "dd", | |
| "details", | |
| "dialog", | |
| "dt", | |
| "figcaption", | |
| "figure", | |
| "footer", | |
| "h1", | |
| "h2", | |
| "h3", | |
| "h4", | |
| "h5", | |
| "h6", | |
| "header", | |
| "label", | |
| "legend", | |
| "li", | |
| "main", | |
| "p", | |
| "section", | |
| "summary", | |
| "td", | |
| "th", | |
| "button", | |
| # <div> is included so that divs used as paragraphs are covered; nested | |
| # divs are naturally excluded by the "leaf block" rule. | |
| "div", | |
| ] | |
| ) | |
| # Inline tags whose content belongs to the surrounding sentence. They are | |
| # replaced with placeholders during block translation and then reinserted. | |
| _INLINE_TAGS = frozenset( | |
| [ | |
| "a", | |
| "abbr", | |
| "acronym", | |
| "b", | |
| "bdo", | |
| "big", | |
| "cite", | |
| "data", | |
| "del", | |
| "dfn", | |
| "em", | |
| "i", | |
| "ins", | |
| "mark", | |
| "q", | |
| "s", | |
| "small", | |
| "span", | |
| "strong", | |
| "sub", | |
| "sup", | |
| "time", | |
| "tt", | |
| "u", | |
| "wbr", | |
| ] | |
| ) | |
| # Regex matching our placeholder tokens | |
| _PLACEHOLDER_RE = re.compile(r"ZT(\d+)ZT") | |
| # Terminal punctuation that standalone inline translations may spuriously add | |
| _TERMINAL_PUNCT_RE = re.compile(r"[.!?]+$") | |
| # --------------------------------------------------------------------------- | |
| # Low-level translation helpers | |
| # --------------------------------------------------------------------------- | |
| def _raw_translate(text, from_code, to_code): | |
| """Thin wrapper around argostranslate with quote normalisation.""" | |
| result = argostranslate.translate.translate(text, from_code, to_code) | |
| # Collapse runs of 2+ consecutive ASCII double-quotes to a single one. | |
| result = re.sub(r'"{2,}', '"', result) | |
| return result | |
| def _translate_inline_text(text, from_code, to_code): | |
| """ | |
| Translate a short inline snippet, stripping any terminal punctuation that | |
| the model may add when given an isolated word or phrase. | |
| """ | |
| translated = _raw_translate(text, from_code, to_code) | |
| # Strip trailing sentence-ending punctuation that doesn't belong inside | |
| # an inline element embedded in a larger sentence. | |
| translated = _TERMINAL_PUNCT_RE.sub("", translated).rstrip() | |
| return translated | |
| # --------------------------------------------------------------------------- | |
| # Block-level translation logic | |
| # --------------------------------------------------------------------------- | |
| def _is_leaf_block(element): | |
| """Return True if *element* contains no block-level descendants.""" | |
| for desc in element.descendants: | |
| if isinstance(desc, Tag) and desc.name in _BLOCK_TAGS: | |
| return False | |
| return True | |
| def _has_translatable_text(element): | |
| """Return True if *element* contains at least one non-whitespace text node | |
| that is not inside a skip tag.""" | |
| for node in element.descendants: | |
| if not isinstance(node, NavigableString) or isinstance(node, Comment): | |
| continue | |
| if not str(node).strip(): | |
| continue | |
| # Check ancestry for skip tags | |
| skip = False | |
| for parent in node.parents: | |
| if parent is element: | |
| break | |
| if isinstance(parent, Tag) and parent.name in _SKIP_TAGS: | |
| skip = True | |
| break | |
| if not skip: | |
| return True | |
| return False | |
| def _translate_inline_element(element, from_code, to_code): | |
| """ | |
| Recursively translate the content of an inline element. | |
| If the element contains only text (no further inline children), translate | |
| the text directly. If it contains nested inline elements, apply the | |
| placeholder approach recursively so nested formatting is preserved. | |
| """ | |
| # Collect direct children | |
| children = list(element.children) | |
| # Check if there are any inline sub-elements | |
| has_nested_inline = any( | |
| isinstance(c, Tag) and c.name in _INLINE_TAGS for c in children | |
| ) | |
| if not has_nested_inline: | |
| # Simple case: translate full text content | |
| inner = element.get_text() | |
| if inner.strip(): | |
| translated = _translate_inline_text(inner.strip(), from_code, to_code) | |
| element.clear() | |
| element.append(NavigableString(translated)) | |
| return | |
| # Check if this element has any *direct* text content (i.e. text not | |
| # coming from a nested inline). When the element is a pure wrapper like | |
| # <b><i>text</i></b> there is no direct text, so building a placeholder | |
| # string of just "ZT0ZT" would be meaningless to translate. In that case, | |
| # fall back to translating the full text content and replacing all children. | |
| has_direct_text = any( | |
| isinstance(c, NavigableString) and not isinstance(c, Comment) and str(c).strip() | |
| for c in children | |
| ) | |
| if not has_direct_text: | |
| inner = element.get_text() | |
| if inner.strip(): | |
| translated = _translate_inline_text(inner.strip(), from_code, to_code) | |
| element.clear() | |
| element.append(NavigableString(translated)) | |
| return | |
| # Recursive case: use placeholder approach on the inline element itself | |
| parts = [] | |
| inline_map = {} | |
| for child in children: | |
| if isinstance(child, NavigableString): | |
| if not isinstance(child, Comment): | |
| parts.append(str(child)) | |
| elif isinstance(child, Tag): | |
| if child.name in _SKIP_TAGS: | |
| parts.append(str(child)) | |
| elif child.name in _INLINE_TAGS: | |
| n = len(inline_map) | |
| inline_map[n] = child | |
| parts.append(f"ZT{n}ZT") | |
| else: | |
| parts.append(element.get_text()) # fallback | |
| raw = "".join(parts).strip() | |
| if not raw: | |
| return | |
| translated_block = _translate_inline_text(raw, from_code, to_code) | |
| for n, child_elem in inline_map.items(): | |
| _translate_inline_element(child_elem, from_code, to_code) | |
| # Reconstruct element | |
| element.clear() | |
| tokens = _PLACEHOLDER_RE.split(translated_block) | |
| for i, tok in enumerate(tokens): | |
| if i % 2 == 0: | |
| if tok: | |
| element.append(NavigableString(tok)) | |
| else: | |
| n = int(tok) | |
| if n in inline_map: | |
| element.append(inline_map[n]) | |
| def _translate_leaf_block(element, from_code, to_code): | |
| """ | |
| Translate a leaf block element using the placeholder strategy. | |
| Direct text children are included verbatim; direct inline-tag children are | |
| replaced with ZT{n}ZT tokens. The resulting string is translated as one | |
| unit (full sentence context). Each inline element is then translated | |
| independently (with terminal punctuation stripped), and the placeholders | |
| are substituted back to reconstruct the element. | |
| If the NMT drops a placeholder (rare), that inline element is silently | |
| omitted from the output rather than producing corrupt HTML. | |
| """ | |
| parts = [] | |
| inline_map = {} # placeholder index -> Tag | |
| for child in list(element.children): | |
| if isinstance(child, Comment): | |
| continue | |
| if isinstance(child, NavigableString): | |
| parts.append(str(child)) | |
| elif isinstance(child, Tag): | |
| if child.name in _SKIP_TAGS: | |
| # Keep skip-tag content literally in the placeholder string so | |
| # it doesn't get sent to the NMT; extract and reattach later. | |
| n = len(inline_map) | |
| inline_map[n] = child | |
| parts.append(f"ZT{n}ZT") | |
| elif child.name in _INLINE_TAGS: | |
| n = len(inline_map) | |
| inline_map[n] = child | |
| parts.append(f"ZT{n}ZT") | |
| else: | |
| # Nested block — shouldn't appear in a leaf block, but if it | |
| # does, include its text inline. | |
| parts.append(child.get_text()) | |
| raw = "".join(parts).strip() | |
| if not raw: | |
| return | |
| # ------------------------------------------------------------------ | |
| # If there are no inline elements, a single translation suffices. | |
| # ------------------------------------------------------------------ | |
| if not inline_map: | |
| translated = _raw_translate(raw, from_code, to_code) | |
| element["title"] = raw | |
| element.clear() | |
| element.append(NavigableString(translated)) | |
| return | |
| # ------------------------------------------------------------------ | |
| # Translate block with placeholders (preserves surrounding context). | |
| # ------------------------------------------------------------------ | |
| translated = _raw_translate(raw, from_code, to_code) | |
| # ------------------------------------------------------------------ | |
| # Sanity: check that placeholders were not significantly displaced by | |
| # the NMT. If the relative position of any placeholder in the output | |
| # differs from its source position by more than 40% of the text length, | |
| # the template translation is unreliable for that placeholder's context | |
| # extraction (the model rearranged the sentence around the placeholder). | |
| # We record which placeholders are "stable" for context extraction. | |
| stable_placeholders = ( | |
| set() | |
| ) # int indices of placeholders not significantly displaced | |
| for n in inline_map: | |
| ph = f"ZT{n}ZT" | |
| src_pos = raw.find(ph) | |
| out_pos = translated.find(ph) | |
| if src_pos == -1 or out_pos == -1: | |
| continue | |
| src_ratio = src_pos / max(len(raw), 1) | |
| out_ratio = out_pos / max(len(translated), 1) | |
| if abs(src_ratio - out_ratio) <= 0.40: | |
| stable_placeholders.add(n) | |
| # ------------------------------------------------------------------ | |
| # Also translate the block with inline content embedded (no placeholders) | |
| # so we can extract context-aware translations for inline elements. | |
| # The user explicitly permits multiple translation passes. | |
| # ------------------------------------------------------------------ | |
| full_parts = [] | |
| for child in list(element.children): | |
| if isinstance(child, Comment): | |
| continue | |
| if isinstance(child, NavigableString): | |
| full_parts.append(str(child)) | |
| elif isinstance(child, Tag): | |
| if child.name not in _SKIP_TAGS: | |
| full_parts.append(child.get_text()) | |
| raw_full = "".join(full_parts).strip() | |
| # Only bother with the second translation when the texts differ | |
| translated_full = ( | |
| _raw_translate(raw_full, from_code, to_code) if raw_full != raw else None | |
| ) | |
| # ------------------------------------------------------------------ | |
| # For each inline element: try context-aware extraction first, then | |
| # fall back to translating the element independently. | |
| # ------------------------------------------------------------------ | |
| for n, child_elem in inline_map.items(): | |
| if child_elem.name in _SKIP_TAGS: | |
| continue # leave skip-tag content untouched | |
| context_text = None | |
| if translated_full and n in stable_placeholders: | |
| # Pass the source inline text's word count as a lower bound so | |
| # we don't accept a single-word fragment (like "The") when the | |
| # NMT rearranged the sentence and the placeholder moved to the front. | |
| src_words = max(1, len(child_elem.get_text().split())) | |
| context_text = _match_context_span( | |
| translated, translated_full, n, min_words=src_words | |
| ) | |
| if context_text is not None: | |
| # Context extraction succeeded: overwrite the element's content | |
| # with the contextually-correct translation. Any nested inline | |
| # formatting is lost here, but context accuracy takes priority. | |
| child_elem.clear() | |
| child_elem.append(NavigableString(context_text)) | |
| else: | |
| # Extraction failed: translate the inline element's content | |
| # independently (works well for short words / phrases). | |
| _translate_inline_element(child_elem, from_code, to_code) | |
| # Reconstruct the element from the translated string + inline elements. | |
| # Store the original plain text as a title attribute so users can hover | |
| # to see the source text. | |
| original_text = raw_full if raw_full else raw | |
| element["title"] = original_text | |
| element.clear() | |
| tokens = _PLACEHOLDER_RE.split(translated) | |
| for i, tok in enumerate(tokens): | |
| if i % 2 == 0: | |
| if tok: | |
| element.append(NavigableString(tok)) | |
| else: | |
| n = int(tok) | |
| if n in inline_map: | |
| element.append(inline_map[n]) | |
| # else: placeholder was dropped by NMT – silently skip | |
| def _match_context_span(template_trans, full_trans, placeholder_n, min_words=1): | |
| """ | |
| Given a translated template string with ``ZT{n}ZT`` placeholders and the | |
| translation of the same block *without* any placeholders, try to extract | |
| the span in *full_trans* that corresponds to placeholder *placeholder_n*. | |
| Uses ``\\b`` word-boundary anchors (so "agrees" does not match inside | |
| "disagrees"). Both left and right boundaries must be located for the | |
| result to be reliable. | |
| *min_words*: minimum word count for the extracted span. Pass the source | |
| inline element's word count to avoid returning a 1-word fragment like | |
| "The" when the NMT moved the placeholder to the sentence start. | |
| Returns the extracted span (stripped), or ``None`` if extraction fails. | |
| """ | |
| tokens = _PLACEHOLDER_RE.split(template_trans) | |
| ph_pos = None | |
| for i in range(1, len(tokens), 2): | |
| if int(tokens[i]) == placeholder_n: | |
| ph_pos = i | |
| break | |
| if ph_pos is None: | |
| return None | |
| left_ctx = tokens[ph_pos - 1] if ph_pos > 0 else "" | |
| right_ctx = tokens[ph_pos + 1] if ph_pos + 1 < len(tokens) else "" | |
| full_lower = full_trans.lower() | |
| def _seq_end(words, haystack, start=0): | |
| # Strip terminal punctuation so "headline." doesn't break \b matching | |
| clean = [re.sub(r"[.,!?;:]+$", "", w) for w in words] | |
| clean = [w for w in clean if w] | |
| if not clean: | |
| return -1 | |
| anchor = r"\s+".join(re.escape(w) for w in clean) | |
| m = re.search(r"\b" + anchor + r"\b", haystack[start:], re.IGNORECASE) | |
| return start + m.end() if m else -1 | |
| def _seq_start(words, haystack, start=0): | |
| # Strip terminal punctuation so "headline." doesn't break \b matching | |
| clean = [re.sub(r"[.,!?;:]+$", "", w) for w in words] | |
| clean = [w for w in clean if w] | |
| if not clean: | |
| return -1 | |
| anchor = r"\s+".join(re.escape(w) for w in clean) | |
| m = re.search(r"\b" + anchor + r"\b", haystack[start:], re.IGNORECASE) | |
| return start + m.start() if m else -1 | |
| # ----- Left boundary ------------------------------------------------ | |
| left_end = None | |
| if not left_ctx.strip(): | |
| left_end = 0 # placeholder is at the template start | |
| else: | |
| left_words = left_ctx.strip().split() | |
| for n_words in range(min(4, len(left_words)), 0, -1): | |
| pos = _seq_end(left_words[-n_words:], full_lower) | |
| if pos != -1: | |
| while pos < len(full_trans) and full_trans[pos] == " ": | |
| pos += 1 | |
| left_end = pos | |
| break | |
| # ----- Right boundary ----------------------------------------------- | |
| right_start = None | |
| if not right_ctx.strip(): | |
| right_start = len(full_trans) # placeholder is at the template end | |
| else: | |
| right_words = right_ctx.strip().split() | |
| search_from = left_end if left_end is not None else 0 | |
| for n_words in range(min(4, len(right_words)), 0, -1): | |
| pos = _seq_start(right_words[:n_words], full_lower, search_from) | |
| if pos != -1: | |
| while pos > search_from and full_trans[pos - 1] == " ": | |
| pos -= 1 | |
| right_start = pos | |
| break | |
| if left_end is None or right_start is None: | |
| return None | |
| if left_end >= right_start: | |
| return None | |
| # When the placeholder was at the template start (left_end=0) and the | |
| # right boundary lands in the first 20% of the full translation, the NMT | |
| # likely rearranged the sentence — the span would be just an article or | |
| # similar fragment, so reject. | |
| if left_end == 0 and not left_ctx.strip(): | |
| if right_start < len(full_trans) * 0.20: | |
| return None | |
| span = full_trans[left_end:right_start].strip() | |
| if not span: | |
| return None | |
| if len(span) > len(full_trans) * 0.80: | |
| return None | |
| if len(span.split()) < min_words: | |
| return None | |
| return span | |
| def _should_translate_node(node): | |
| """ | |
| Return True when a NavigableString carries translatable human text | |
| (used for the fallback per-node path). | |
| """ | |
| if isinstance(node, Comment): | |
| return False | |
| text = str(node) | |
| if not text.strip(): | |
| return False | |
| for parent in node.parents: | |
| if hasattr(parent, "name") and parent.name in _SKIP_TAGS: | |
| return False | |
| return True | |
| def translate_html_string(html_content, from_code, to_code): | |
| """ | |
| Parse *html_content*, translate every visible text node, and return the | |
| modified HTML as a string. | |
| Block-level elements are translated as complete units (preserving inline | |
| markup via placeholders). Any remaining orphan text nodes are translated | |
| individually as a fallback. | |
| """ | |
| soup = BeautifulSoup(html_content, "lxml") | |
| # ------------------------------------------------------------------ | |
| # Pass 1: translate leaf block elements as whole units | |
| # ------------------------------------------------------------------ | |
| # Collect leaf blocks first to avoid mutating the tree while iterating. | |
| leaf_blocks = [ | |
| elem | |
| for elem in soup.find_all(_BLOCK_TAGS) | |
| if _is_leaf_block(elem) and _has_translatable_text(elem) | |
| ] | |
| # Track which text nodes have already been translated so the fallback | |
| # pass doesn't double-translate them. | |
| translated_blocks = set() | |
| total_blocks = len(leaf_blocks) | |
| print(f"Found {total_blocks} translatable leaf-block elements.") | |
| for idx, elem in enumerate(leaf_blocks, start=1): | |
| try: | |
| _translate_leaf_block(elem, from_code, to_code) | |
| translated_blocks.add(id(elem)) | |
| except Exception as e: | |
| print(f" [{idx}/{total_blocks}] Warning: block translation error: {e}") | |
| if idx % 50 == 0 or idx == total_blocks: | |
| print(f" Translated {idx}/{total_blocks} blocks...") | |
| # ------------------------------------------------------------------ | |
| # Pass 2: fallback – translate any remaining orphan text nodes that | |
| # were not covered by a leaf-block ancestor. | |
| # ------------------------------------------------------------------ | |
| orphan_nodes = [] | |
| for node in soup.find_all(string=True): | |
| if not _should_translate_node(node): | |
| continue | |
| # Check if any ancestor was a translated leaf block | |
| covered = False | |
| for parent in node.parents: | |
| if id(parent) in translated_blocks: | |
| covered = True | |
| break | |
| if not covered: | |
| orphan_nodes.append(node) | |
| if orphan_nodes: | |
| print(f"Found {len(orphan_nodes)} orphan text nodes (fallback).") | |
| for idx, node in enumerate(orphan_nodes, start=1): | |
| original = str(node) | |
| clean = original.strip() | |
| if not clean: | |
| continue | |
| try: | |
| translated = _raw_translate(clean, from_code, to_code) | |
| except Exception as e: | |
| print(f" Warning: skipping orphan node (translation error): {e}") | |
| continue | |
| leading = original[: len(original) - len(original.lstrip())] | |
| trailing = original[len(original.rstrip()) :] | |
| node.replace_with(NavigableString(leading + translated + trailing)) | |
| return str(soup) | |
| def translate_html_file(input_path, output_path, from_code, to_code): | |
| """ | |
| Read *input_path*, translate its text nodes, and write to *output_path*. | |
| """ | |
| install_local_or_download_model(from_code, to_code) | |
| print(f"Reading: {input_path}") | |
| with open(input_path, "r", encoding="utf-8", errors="replace") as fh: | |
| html_content = fh.read() | |
| print(f"Translating {from_code} -> {to_code} ...") | |
| translated_html = translate_html_string(html_content, from_code, to_code) | |
| with open(output_path, "w", encoding="utf-8") as fh: | |
| fh.write(translated_html) | |
| print(f"Done! Saved to: {output_path}") | |
| def main(): | |
| parser = argparse.ArgumentParser( | |
| description="Translate an HTML file at the DOM level using Argos Translate.", | |
| ) | |
| parser.add_argument( | |
| "input_file", | |
| nargs="?", | |
| help="Path to the input HTML file.", | |
| ) | |
| parser.add_argument( | |
| "--from", | |
| dest="from_lang", | |
| default="ru", | |
| help="Source language code (default: ru).", | |
| ) | |
| parser.add_argument( | |
| "--to", | |
| dest="to_lang", | |
| default="en", | |
| help="Target language code (default: en).", | |
| ) | |
| parser.add_argument( | |
| "--output", | |
| dest="output_file", | |
| default=None, | |
| help=( | |
| "Path for the translated HTML output. " | |
| "Defaults to <input_stem>_translated.html next to the input file." | |
| ), | |
| ) | |
| parser.add_argument( | |
| "--language-codes", | |
| action="store_true", | |
| help="List language abbreviations and Argos translation pairs, then exit.", | |
| ) | |
| args = parser.parse_args() | |
| if args.language_codes: | |
| list_language_codes() | |
| return | |
| if not args.input_file: | |
| parser.error( | |
| "the following arguments are required: input_file " | |
| "(unless --language-codes is used)" | |
| ) | |
| input_file = args.input_file | |
| if args.output_file: | |
| output_file = args.output_file | |
| else: | |
| stem, _ = os.path.splitext(input_file) | |
| output_file = stem + "_translated.html" | |
| translate_html_file( | |
| input_path=input_file, | |
| output_path=output_file, | |
| from_code=args.from_lang, | |
| to_code=args.to_lang, | |
| ) | |
| if __name__ == "__main__": | |
| main() |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| import os | |
| import sys | |
| import shutil | |
| import argparse | |
| import subprocess | |
| import tempfile | |
| import fitz # PyMuPDF | |
| import argostranslate.translate | |
| from translate_common import ( | |
| install_local_or_download_model, | |
| list_language_codes, | |
| ) | |
| def find_ghostscript_executable(): | |
| """ | |
| Find Ghostscript executable across platforms. | |
| """ | |
| candidates = ["gs", "gswin64c", "gswin32c"] | |
| for c in candidates: | |
| path = shutil.which(c) | |
| if path: | |
| return path | |
| return None | |
| def run_ghostscript_compress(input_pdf, output_pdf, preset="ebook"): | |
| """ | |
| Compress PDF with Ghostscript. | |
| Returns True on success, False on failure. | |
| """ | |
| gs_exe = find_ghostscript_executable() | |
| if not gs_exe: | |
| print("Warning: Ghostscript not found (gs). Skipping GS compression.") | |
| return False | |
| cmd = [ | |
| gs_exe, | |
| "-sDEVICE=pdfwrite", | |
| "-dCompatibilityLevel=1.6", | |
| "-dNOPAUSE", | |
| "-dQUIET", | |
| "-dBATCH", | |
| "-dDetectDuplicateImages=true", | |
| "-dCompressFonts=true", | |
| ] | |
| if preset != "default": | |
| cmd.append(f"-dPDFSETTINGS=/{preset}") | |
| cmd.append(f"-sOutputFile={output_pdf}") | |
| cmd.append(input_pdf) | |
| print(f"Running Ghostscript compression (preset: {preset}) ...") | |
| try: | |
| subprocess.run(cmd, check=True) | |
| return True | |
| except subprocess.CalledProcessError as e: | |
| print(f"Warning: Ghostscript compression failed: {e}") | |
| return False | |
| def translate_and_overlay_blocks( | |
| input_path, | |
| output_path, | |
| from_code, | |
| to_code, | |
| keep_original, | |
| use_ghostscript=True, | |
| gs_preset="ebook", | |
| ): | |
| install_local_or_download_model(from_code, to_code) | |
| doc = fitz.open(input_path) | |
| print(f"Processing {len(doc)} pages...") | |
| for page_num, page in enumerate(doc): | |
| print(f"Translating page {page_num + 1}/{len(doc)}...") | |
| blocks = page.get_text("blocks") | |
| translated_blocks = [] # list of (rect, translated_text) | |
| # 1) Collect translations and (optionally) redaction annotations | |
| for block in blocks: | |
| if len(block) < 7: | |
| continue | |
| x0, y0, x1, y1, text, _block_no, block_type = block[:7] | |
| if block_type != 0: | |
| continue | |
| if not text or not text.strip(): | |
| continue | |
| clean_text = text.replace("\n", " ").strip() | |
| if len(clean_text) <= 1: | |
| continue | |
| try: | |
| translated_text = argostranslate.translate.translate( | |
| clean_text, from_code, to_code | |
| ) | |
| except Exception as e: | |
| print(f" Skipping block (translation error): {e}") | |
| continue | |
| rect = fitz.Rect(x0, y0, x1, y1) | |
| translated_blocks.append((rect, translated_text)) | |
| if not keep_original: | |
| page.add_redact_annot(rect, fill=(1, 1, 1)) | |
| # 2) Apply redactions ONCE per page | |
| if not keep_original: | |
| try: | |
| # Keep images unchanged to reduce bloat from redaction processing | |
| try: | |
| page.apply_redactions(images=0) | |
| except TypeError: | |
| page.apply_redactions() | |
| except Exception as e: | |
| print(f" Warning: apply_redactions failed on page {page_num + 1}: {e}") | |
| # 3) Insert translated text | |
| for rect, translated_text in translated_blocks: | |
| font_size = 11.0 | |
| res = -1.0 | |
| while res < 0 and font_size > 5: | |
| res = page.insert_textbox( | |
| rect, | |
| translated_text, | |
| fontsize=font_size, | |
| fontname="helv", | |
| color=(0, 0, 0), | |
| align=0, | |
| ) | |
| if res < 0: | |
| font_size -= 0.5 | |
| # Save intermediate translated PDF (optimized PyMuPDF save) | |
| tmp_fd, tmp_pdf_path = tempfile.mkstemp(suffix=".pdf") | |
| os.close(tmp_fd) | |
| save_kwargs = dict( | |
| garbage=4, | |
| deflate=True, | |
| clean=True, | |
| incremental=False, | |
| ) | |
| try: | |
| try: | |
| doc.save( | |
| tmp_pdf_path, | |
| **save_kwargs, | |
| deflate_images=True, | |
| deflate_fonts=True, | |
| ) | |
| except TypeError: | |
| doc.save(tmp_pdf_path, **save_kwargs) | |
| finally: | |
| doc.close() | |
| # Optional Ghostscript final compression | |
| if use_ghostscript: | |
| ok = run_ghostscript_compress(tmp_pdf_path, output_path, preset=gs_preset) | |
| if not ok: | |
| # Fallback: keep PyMuPDF output | |
| shutil.move(tmp_pdf_path, output_path) | |
| print("Saved without Ghostscript compression due to GS issue.") | |
| else: | |
| try: | |
| os.remove(tmp_pdf_path) | |
| except OSError: | |
| pass | |
| else: | |
| shutil.move(tmp_pdf_path, output_path) | |
| print(f"Done! Saved to: {output_path}") | |
| def main(): | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument( | |
| "input_file", | |
| nargs="?", | |
| help="Path to the input PDF file.", | |
| ) | |
| parser.add_argument( | |
| "--from", | |
| dest="from_lang", | |
| default="ru", | |
| help="Source language code (default: ru).", | |
| ) | |
| parser.add_argument( | |
| "--to", | |
| dest="to_lang", | |
| default="en", | |
| help="Target language code (default: en).", | |
| ) | |
| parser.add_argument( | |
| "--keep-original", | |
| action="store_true", | |
| help="Keep original text layer beneath translation (no redaction).", | |
| ) | |
| parser.add_argument( | |
| "--language-codes", | |
| action="store_true", | |
| help="List language abbreviations and Argos translation pairs, then exit.", | |
| ) | |
| parser.add_argument( | |
| "--no-gs", | |
| action="store_true", | |
| help="Disable Ghostscript compression pass.", | |
| ) | |
| parser.add_argument( | |
| "--gs-preset", | |
| choices=["screen", "ebook", "printer", "prepress", "default"], | |
| default="ebook", | |
| help="Ghostscript PDFSETTINGS preset (default: ebook).", | |
| ) | |
| args = parser.parse_args() | |
| if args.language_codes: | |
| list_language_codes() | |
| return | |
| if not args.input_file: | |
| parser.error( | |
| "the following arguments are required: input_file (unless --language-codes is used)" | |
| ) | |
| input_file = args.input_file | |
| output_file = os.path.splitext(input_file)[0] + "_translated.pdf" | |
| translate_and_overlay_blocks( | |
| input_path=input_file, | |
| output_path=output_file, | |
| from_code=args.from_lang, | |
| to_code=args.to_lang, | |
| keep_original=args.keep_original, | |
| use_ghostscript=not args.no_gs, | |
| gs_preset=args.gs_preset, | |
| ) | |
| if __name__ == "__main__": | |
| main() |
Author
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
Updated script, and made it installable by
uv tool install:Get
uvfrom here: https://docs.astral.sh/uv/getting-started/installation/If you have Ghostscript on your machine, drop the
--no-gsfor considerable size reductions.