Import AITURK IDE 1.0.0-beta.1 from Hermes 63279301; preserve MIT license
This commit is contained in:
@@ -0,0 +1,156 @@
|
||||
#!/usr/bin/env python3
|
||||
# MIT License. Part of the Hermes docx skill.
|
||||
"""Health-check a .docx package and report issues as JSON.
|
||||
|
||||
Usage: docx_validate.py file.docx
|
||||
|
||||
Checks (health-check tier, NOT full XSD schema validation):
|
||||
- the file is a readable zip and python-docx can open it
|
||||
- required package parts exist ([Content_Types].xml, document.xml)
|
||||
- every relationship in every .rels file resolves to a part in the
|
||||
package (dangling image/hyperlink/etc. rels are reported; external
|
||||
targets such as hyperlinks are skipped)
|
||||
- r:embed / r:id references in document.xml resolve to relationships
|
||||
- embedded images are non-empty and start with known magic bytes
|
||||
(PNG/JPEG/GIF/BMP/TIFF/EMF/WMF/SVG); no PIL required
|
||||
- paragraph and run style ids referenced by the document exist in
|
||||
styles.xml
|
||||
|
||||
Output: {"ok": bool, "issues": [{"severity": "error"|"warning", ...}]}
|
||||
Exit code 1 when any error-severity issue is found (warnings exit 0).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import posixpath
|
||||
import sys
|
||||
import zipfile
|
||||
|
||||
from lxml import etree
|
||||
|
||||
W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
||||
R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
|
||||
PR = "http://schemas.openxmlformats.org/package/2006/relationships"
|
||||
|
||||
IMAGE_MAGIC = (
|
||||
b"\x89PNG\r\n\x1a\n", b"\xff\xd8\xff", b"GIF87a", b"GIF89a",
|
||||
b"BM", b"II*\x00", b"MM\x00*",
|
||||
b"\x01\x00\x00\x00", # EMF
|
||||
b"\xd7\xcd\xc6\x9a", b"\x01\x00\x09\x00", # WMF variants
|
||||
b"<?xml", b"<svg",
|
||||
)
|
||||
|
||||
|
||||
def _issue(issues, severity, code, detail):
|
||||
issues.append({"severity": severity, "code": code, "detail": detail})
|
||||
|
||||
|
||||
def _rel_target(base_part: str, target: str) -> str:
|
||||
base_dir = posixpath.dirname(base_part)
|
||||
return posixpath.normpath(posixpath.join(base_dir, target)).lstrip("/")
|
||||
|
||||
|
||||
def validate(path: str) -> dict:
|
||||
issues: list[dict] = []
|
||||
|
||||
try:
|
||||
zf = zipfile.ZipFile(path)
|
||||
except (OSError, zipfile.BadZipFile) as exc:
|
||||
_issue(issues, "error", "not-a-zip", str(exc))
|
||||
return {"ok": False, "issues": issues}
|
||||
|
||||
names = set(zf.namelist())
|
||||
bad = zf.testzip()
|
||||
if bad is not None:
|
||||
_issue(issues, "error", "corrupt-member", f"CRC check failed: {bad}")
|
||||
|
||||
for required in ("[Content_Types].xml", "word/document.xml"):
|
||||
if required not in names:
|
||||
_issue(issues, "error", "missing-part",
|
||||
f"required part absent: {required}")
|
||||
if issues and any(i["severity"] == "error" for i in issues):
|
||||
return {"ok": False, "issues": issues}
|
||||
|
||||
# --- relationships resolve ------------------------------------------
|
||||
rel_ids_by_source: dict[str, dict] = {}
|
||||
for rels_name in [n for n in names if n.endswith(".rels")]:
|
||||
try:
|
||||
root = etree.fromstring(zf.read(rels_name))
|
||||
except etree.XMLSyntaxError as exc:
|
||||
_issue(issues, "error", "bad-rels-xml", f"{rels_name}: {exc}")
|
||||
continue
|
||||
source_part = posixpath.normpath(
|
||||
posixpath.join(posixpath.dirname(rels_name), ".."))
|
||||
source_part = "" if source_part == "." else source_part
|
||||
ids = {}
|
||||
for rel in root.iter(f"{{{PR}}}Relationship"):
|
||||
rid, target = rel.get("Id"), rel.get("Target", "")
|
||||
mode = rel.get("TargetMode", "Internal")
|
||||
ids[rid] = target
|
||||
if mode == "External":
|
||||
continue
|
||||
resolved = _rel_target(source_part + "/x" if source_part
|
||||
else "x", target)
|
||||
if resolved not in names:
|
||||
_issue(issues, "error", "dangling-rel",
|
||||
f"{rels_name}: {rid} -> {target} (missing part)")
|
||||
rel_ids_by_source[source_part or "_package"] = ids
|
||||
|
||||
# --- r:id / r:embed references in document.xml -----------------------
|
||||
doc_root = etree.fromstring(zf.read("word/document.xml"))
|
||||
doc_rels = rel_ids_by_source.get("word", {})
|
||||
for el in doc_root.iter():
|
||||
for attr in (f"{{{R}}}id", f"{{{R}}}embed", f"{{{R}}}link"):
|
||||
rid = el.get(attr)
|
||||
if rid and rid not in doc_rels:
|
||||
_issue(issues, "error", "unresolved-reference",
|
||||
f"document.xml references {rid} with no relationship")
|
||||
|
||||
# --- embedded images decode ------------------------------------------
|
||||
for name in [n for n in names if n.startswith("word/media/")]:
|
||||
data = zf.read(name)
|
||||
if not data:
|
||||
_issue(issues, "error", "empty-image", name)
|
||||
elif not any(data.startswith(m) for m in IMAGE_MAGIC):
|
||||
_issue(issues, "warning", "unknown-image-format",
|
||||
f"{name}: unrecognized magic bytes")
|
||||
|
||||
# --- styles referenced exist ------------------------------------------
|
||||
defined = set()
|
||||
if "word/styles.xml" in names:
|
||||
styles_root = etree.fromstring(zf.read("word/styles.xml"))
|
||||
defined = {s.get(f"{{{W}}}styleId")
|
||||
for s in styles_root.iter(f"{{{W}}}style")}
|
||||
for tag, attr in ((f"{{{W}}}pStyle", f"{{{W}}}val"),
|
||||
(f"{{{W}}}rStyle", f"{{{W}}}val"),
|
||||
(f"{{{W}}}tblStyle", f"{{{W}}}val")):
|
||||
for el in doc_root.iter(tag):
|
||||
sid = el.get(attr)
|
||||
if sid and sid not in defined:
|
||||
_issue(issues, "error", "missing-style",
|
||||
f"style id referenced but not defined: {sid}")
|
||||
|
||||
# --- python-docx can open it ------------------------------------------
|
||||
try:
|
||||
from docx import Document
|
||||
Document(path)
|
||||
except Exception as exc: # noqa: BLE001 - triage tool, report anything
|
||||
_issue(issues, "error", "python-docx-open-failed", str(exc))
|
||||
|
||||
ok = not any(i["severity"] == "error" for i in issues)
|
||||
return {"ok": ok, "issues": issues}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(
|
||||
description="Health-check a .docx (not XSD schema validation).")
|
||||
ap.add_argument("path", help="the .docx file to check")
|
||||
args = ap.parse_args()
|
||||
report = validate(args.path)
|
||||
print(json.dumps(report, ensure_ascii=False))
|
||||
return 0 if report["ok"] else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user