| Server IP : 91.134.83.25 / Your IP : 216.73.216.79 Web Server : Apache System : Linux plesk.serveurapc.fr 6.1.0-51-amd64 #1 SMP PREEMPT_DYNAMIC Debian 6.1.177-1 (2026-07-16) x86_64 User : marrasse ( 10057) PHP Version : 8.2.32 Disable Function : opcache_get_status MySQL : OFF | cURL : ON | WGET : OFF | Perl : OFF | Python : OFF | Sudo : OFF | Pkexec : OFF Directory : /var/www/vhosts/as-cp.fr/.claude/remote/plugins/f5be4c412689f929/skills/docx/scripts/ |
Upload File : |
"""Merge adjacent identically-formatted runs in a DOCX.
Word fragments paragraph text across many <w:r> elements (revision ids,
spell-check markers, editing history), which makes find-and-replace on
word/document.xml unreliable — the string you're looking for is split
across runs. This coalesces adjacent runs whose formatting (<w:rPr>) is
identical, strips rsid attributes and proofErr markers, and consolidates the
text elements — <w:t>, and <w:delText> for text inside a tracked deletion.
Rendering is unchanged. The text you search is what Word draws, which is not
always the bytes in the file: an element without xml:space="preserve" has its
edge whitespace trimmed before it reaches the page, so `<w:t>Hello </w:t>`
followed by `<w:t>world</w:t>` reads "Helloworld" and merges to exactly that.
Runs in two different <w:ins>/<w:del> wrappers are never merged: that would
rewrite tracked-change structure, collapsing separate revisions into one.
Only word/document.xml is processed (not headers, footers, or footnotes).
Usage:
python merge_runs.py unpacked/ # after unzip, before editing
python merge_runs.py document.docx # rewrite in place
python merge_runs.py document.docx -o out.docx
"""
import argparse
import sys
import tempfile
import zipfile
from pathlib import Path
import defusedxml.minidom
from office.helpers import XML_SPACE, rendered_text, rezip, safe_extract
WORDML_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
def merge_runs(input_dir: str) -> tuple[int, str]:
doc_xml = Path(input_dir) / "word" / "document.xml"
if not doc_xml.exists():
return 0, f"Error: {doc_xml} not found"
try:
dom = defusedxml.minidom.parseString(doc_xml.read_text(encoding="utf-8"))
root = dom.documentElement
run_names = _run_tag_names(root)
_remove_elements(root, "proofErr")
runs = _find_runs(root, run_names)
_strip_rsid_attrs(runs)
merge_count = 0
for container in {run.parentNode for run in runs}:
merge_count += _merge_runs_in(container, run_names)
doc_xml.write_bytes(dom.toxml(encoding="UTF-8"))
return merge_count, f"Merged {merge_count} runs"
except Exception as e:
return 0, f"Error: {e}"
def _is_element(node, tag: str) -> bool:
name = node.localName or node.tagName
return name == tag or name.endswith(f":{tag}")
def _run_tag_names(root) -> set[str]:
names = set()
for attr in root.attributes.values():
if attr.value == WORDML_NS:
if attr.name == "xmlns":
names.add("r")
elif attr.name.startswith("xmlns:"):
names.add(attr.name.split(":", 1)[1] + ":r")
return names or {"w:r", "r"}
def _find_elements(root, tag: str) -> list:
results = []
def traverse(node):
if node.nodeType == node.ELEMENT_NODE:
if _is_element(node, tag):
results.append(node)
for child in node.childNodes:
traverse(child)
traverse(root)
return results
def _find_runs(root, run_names: set[str]) -> list:
return [e for e in _find_elements(root, "r") if _is_run(e, run_names)]
def _get_child(parent, tag: str):
return next(iter(_get_children(parent, tag)), None)
def _get_children(parent, tag: str) -> list:
return [
child
for child in parent.childNodes
if child.nodeType == child.ELEMENT_NODE and _is_element(child, tag)
]
def _is_adjacent(elem1, elem2) -> bool:
node = elem1.nextSibling
while node:
if node == elem2:
return True
if node.nodeType == node.ELEMENT_NODE:
return False
if node.nodeType == node.TEXT_NODE and node.data.strip(XML_SPACE):
return False
node = node.nextSibling
return False
def _remove_elements(root, tag: str):
for elem in _find_elements(root, tag):
if elem.parentNode:
elem.parentNode.removeChild(elem)
def _strip_rsid_attrs(runs: list):
for run in runs:
for attr in list(run.attributes.values()):
if "rsid" in attr.name.lower():
run.removeAttribute(attr.name)
def _merge_runs_in(container, run_names: set[str]) -> int:
merge_count = 0
run = _first_child_run(container, run_names)
while run:
while True:
next_elem = _next_element_sibling(run)
if next_elem and _is_run(next_elem, run_names) and _can_merge(run, next_elem):
_merge_run_content(run, next_elem)
container.removeChild(next_elem)
merge_count += 1
else:
break
_consolidate_text(run)
run = _next_sibling_run(run, run_names)
return merge_count
def _first_child_run(container, run_names: set[str]):
for child in container.childNodes:
if child.nodeType == child.ELEMENT_NODE and _is_run(child, run_names):
return child
return None
def _next_element_sibling(node):
sibling = node.nextSibling
while sibling:
if sibling.nodeType == sibling.ELEMENT_NODE:
return sibling
sibling = sibling.nextSibling
return None
def _next_sibling_run(node, run_names: set[str]):
sibling = node.nextSibling
while sibling:
if sibling.nodeType == sibling.ELEMENT_NODE:
if _is_run(sibling, run_names):
return sibling
sibling = sibling.nextSibling
return None
def _is_run(node, run_names: set[str]) -> bool:
return node.tagName in run_names
def _can_merge(run1, run2) -> bool:
rpr1 = _get_child(run1, "rPr")
rpr2 = _get_child(run2, "rPr")
if (rpr1 is None) != (rpr2 is None):
return False
if rpr1 is None:
return True
return rpr1.toxml() == rpr2.toxml()
def _merge_run_content(target, source):
for child in list(source.childNodes):
if child.nodeType == child.ELEMENT_NODE:
name = child.localName or child.tagName
if name != "rPr" and not name.endswith(":rPr"):
target.appendChild(child)
def _element_text(elem) -> str:
return "".join(
child.data
for child in elem.childNodes
if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE)
)
def _has_preserve(elem) -> bool:
return elem.getAttribute("xml:space") == "preserve"
def _rendered_text(elem) -> str:
return rendered_text(_element_text(elem), _has_preserve(elem))
def _consolidate_text(run):
for tag in ("t", "delText"):
_consolidate_text_elements(run, tag)
def _consolidate_text_elements(run, tag: str):
t_elements = _get_children(run, tag)
for i in range(len(t_elements) - 1, 0, -1):
curr, prev = t_elements[i], t_elements[i - 1]
if _is_adjacent(prev, curr):
merged = _rendered_text(prev) + _rendered_text(curr)
had_preserve = _has_preserve(prev) or _has_preserve(curr)
new_text = run.ownerDocument.createTextNode(merged)
for node in list(prev.childNodes):
if node.nodeType in (node.TEXT_NODE, node.CDATA_SECTION_NODE):
prev.removeChild(node)
else:
run.insertBefore(node, curr)
prev.appendChild(new_text)
for node in list(curr.childNodes):
if node.nodeType not in (node.TEXT_NODE, node.CDATA_SECTION_NODE):
run.insertBefore(node, curr)
if merged != merged.strip(XML_SPACE) or had_preserve:
prev.setAttribute("xml:space", "preserve")
elif prev.hasAttribute("xml:space"):
prev.removeAttribute("xml:space")
run.removeChild(curr)
def _merge_or_die(path: Path) -> str:
_, msg = merge_runs(str(path))
if msg.startswith("Error"):
print(msg, file=sys.stderr)
sys.exit(1)
return msg
def main() -> None:
p = argparse.ArgumentParser(
description="Merge adjacent identically-formatted runs in a DOCX (directory or .docx file)."
)
p.add_argument("input", help="Unpacked DOCX directory OR a .docx/.dotx file")
p.add_argument(
"-o", "--output",
help="Output .docx path (only valid when input is a .docx; default: overwrite input)",
)
args = p.parse_args()
src = Path(args.input)
try:
if src.is_dir():
if args.output:
p.error("--output is only valid for .docx input; directory input is modified in place")
print(_merge_or_die(src))
elif src.is_file() and src.suffix.lower() in (".docx", ".dotx"):
out = Path(args.output) if args.output else src
with tempfile.TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
with zipfile.ZipFile(src) as zf:
safe_extract(zf, tmp_path)
msg = _merge_or_die(tmp_path)
rezip(tmp_path, out)
print(f"{msg}; wrote {out}")
else:
print(f"Error: {src} is neither a directory nor a .docx/.dotx file", file=sys.stderr)
sys.exit(1)
except (OSError, ValueError, zipfile.BadZipFile) as e:
print(f"Error: {e}", file=sys.stderr)
sys.exit(1)
if __name__ == "__main__":
main()