docx

Create, read, edit, template, and review Word .docx files.

  • word
  • docx
  • documents
  • office
  • templates
  • revisions
  • comments

Declared platforms: linux · macos · windows

Install
npx skills add 'https://github.com/NousResearch/hermes-agent/tree/main/skills/productivity/docx'
Download bundle ↓
main · 24fd22bScanned 2026-09-15

Contributors

GitHub-linked commit authors for this SKILL.md at the saved revision. Co-authors and history before file renames are not included.

File history ↗
View on GitHub
← Back to SKILL.md
#!/usr/bin/env python3# MIT License. Part of the Hermes docx skill."""Health-check a .docx package and report issues as JSON. Usage: docx_validate.py file.docx Checks (health-check tier, NOT full XSD schema validation):  - the file is a readable zip and python-docx can open it  - required package parts exist ([Content_Types].xml, document.xml)  - every relationship in every .rels file resolves to a part in the    package (dangling image/hyperlink/etc. rels are reported; external    targets such as hyperlinks are skipped)  - r:embed / r:id references in document.xml resolve to relationships  - embedded images are non-empty and start with known magic bytes    (PNG/JPEG/GIF/BMP/TIFF/EMF/WMF/SVG); no PIL required  - paragraph and run style ids referenced by the document exist in    styles.xml Output: {"ok": bool, "issues": [{"severity": "error"|"warning", ...}]}Exit code 1 when any error-severity issue is found (warnings exit 0)."""from __future__ import annotations import argparseimport jsonimport posixpathimport sysimport zipfile from lxml import etree W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"PR = "http://schemas.openxmlformats.org/package/2006/relationships" IMAGE_MAGIC = (    b"\x89PNG\r\n\x1a\n", b"\xff\xd8\xff", b"GIF87a", b"GIF89a",    b"BM", b"II*\x00", b"MM\x00*",    b"\x01\x00\x00\x00",              # EMF    b"\xd7\xcd\xc6\x9a", b"\x01\x00\x09\x00",  # WMF variants    b"<?xml", b"<svg",)  def _issue(issues, severity, code, detail):    issues.append({"severity": severity, "code": code, "detail": detail})  def _rel_target(base_part: str, target: str) -> str:    base_dir = posixpath.dirname(base_part)    return posixpath.normpath(posixpath.join(base_dir, target)).lstrip("/")  def validate(path: str) -> dict:    issues: list[dict] = []     try:        zf = zipfile.ZipFile(path)    except (OSError, zipfile.BadZipFile) as exc:        _issue(issues, "error", "not-a-zip", str(exc))        return {"ok": False, "issues": issues}     names = set(zf.namelist())    bad = zf.testzip()    if bad is not None:        _issue(issues, "error", "corrupt-member", f"CRC check failed: {bad}")     for required in ("[Content_Types].xml", "word/document.xml"):        if required not in names:            _issue(issues, "error", "missing-part",                   f"required part absent: {required}")    if issues and any(i["severity"] == "error" for i in issues):        return {"ok": False, "issues": issues}     # --- relationships resolve ------------------------------------------    rel_ids_by_source: dict[str, dict] = {}    for rels_name in [n for n in names if n.endswith(".rels")]:        try:            root = etree.fromstring(zf.read(rels_name))        except etree.XMLSyntaxError as exc:            _issue(issues, "error", "bad-rels-xml", f"{rels_name}: {exc}")            continue        source_part = posixpath.normpath(            posixpath.join(posixpath.dirname(rels_name), ".."))        source_part = "" if source_part == "." else source_part        ids = {}        for rel in root.iter(f"{{{PR}}}Relationship"):            rid, target = rel.get("Id"), rel.get("Target", "")            mode = rel.get("TargetMode", "Internal")            ids[rid] = target            if mode == "External":                continue            resolved = _rel_target(source_part + "/x" if source_part                                   else "x", target)            if resolved not in names:                _issue(issues, "error", "dangling-rel",                       f"{rels_name}: {rid} -> {target} (missing part)")        rel_ids_by_source[source_part or "_package"] = ids     # --- r:id / r:embed references in document.xml -----------------------    doc_root = etree.fromstring(zf.read("word/document.xml"))    doc_rels = rel_ids_by_source.get("word", {})    for el in doc_root.iter():        for attr in (f"{{{R}}}id", f"{{{R}}}embed", f"{{{R}}}link"):            rid = el.get(attr)            if rid and rid not in doc_rels:                _issue(issues, "error", "unresolved-reference",                       f"document.xml references {rid} with no relationship")     # --- embedded images decode ------------------------------------------    for name in [n for n in names if n.startswith("word/media/")]:        data = zf.read(name)        if not data:            _issue(issues, "error", "empty-image", name)        elif not any(data.startswith(m) for m in IMAGE_MAGIC):            _issue(issues, "warning", "unknown-image-format",                   f"{name}: unrecognized magic bytes")     # --- styles referenced exist ------------------------------------------    defined = set()    if "word/styles.xml" in names:        styles_root = etree.fromstring(zf.read("word/styles.xml"))        defined = {s.get(f"{{{W}}}styleId")                   for s in styles_root.iter(f"{{{W}}}style")}    for tag, attr in ((f"{{{W}}}pStyle", f"{{{W}}}val"),                      (f"{{{W}}}rStyle", f"{{{W}}}val"),                      (f"{{{W}}}tblStyle", f"{{{W}}}val")):        for el in doc_root.iter(tag):            sid = el.get(attr)            if sid and sid not in defined:                _issue(issues, "error", "missing-style",                       f"style id referenced but not defined: {sid}")     # --- python-docx can open it ------------------------------------------    try:        from docx import Document        Document(path)    except Exception as exc:  # noqa: BLE001 - triage tool, report anything        _issue(issues, "error", "python-docx-open-failed", str(exc))     ok = not any(i["severity"] == "error" for i in issues)    return {"ok": ok, "issues": issues}  def main() -> int:    ap = argparse.ArgumentParser(        description="Health-check a .docx (not XSD schema validation).")    ap.add_argument("path", help="the .docx file to check")    args = ap.parse_args()    report = validate(args.path)    print(json.dumps(report, ensure_ascii=False))    return 0 if report["ok"] else 1  if __name__ == "__main__":    sys.exit(main()) 
Referenced from SKILL.md