Import AITURK IDE 1.0.0-beta.1 from Hermes 63279301; preserve MIT license

This commit is contained in:
2026-09-05 13:26:46 +03:00
commit 03634b1ca3
11340 changed files with 3442369 additions and 0 deletions
@@ -0,0 +1,525 @@
# MIT License. End-to-end tests for the docx skill.
"""Pytest suite proving create / read / edit / template round-trips.
Runs the scripts as subprocesses (argparse CLIs) and also verifies the
outputs with python-docx directly. Stdlib + python-docx only; all
fixtures are generated on the fly; no network.
"""
from __future__ import annotations
import json
import os
import struct
import subprocess
import sys
import zlib
from pathlib import Path
import pytest
from docx import Document
SKILL = Path(__file__).resolve().parent.parent
SCRIPTS = SKILL / "scripts"
NON_ASCII = "Фамилия — test"
def make_png(path: Path) -> None:
"""Write a tiny valid 2x2 red PNG using only stdlib."""
def chunk(tag: bytes, data: bytes) -> bytes:
return (struct.pack(">I", len(data)) + tag + data
+ struct.pack(">I", zlib.crc32(tag + data) & 0xFFFFFFFF))
ihdr = struct.pack(">IIBBBBB", 2, 2, 8, 2, 0, 0, 0)
raw = b"".join(b"\x00" + b"\xff\x00\x00" * 2 for _ in range(2))
png = (b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", ihdr)
+ chunk(b"IDAT", zlib.compress(raw)) + chunk(b"IEND", b""))
path.write_bytes(png)
def run(script: str, *args: str) -> dict:
env = dict(os.environ)
env["LC_ALL"] = "C" # prove no locale-default text reads
env["PYTHONIOENCODING"] = "utf-8"
proc = subprocess.run(
[sys.executable, str(SCRIPTS / script), *map(str, args)],
capture_output=True, env=env)
assert proc.returncode == 0, proc.stderr.decode("utf-8", "replace")
return json.loads(proc.stdout.decode("utf-8"))
@pytest.fixture(scope="module")
def workdir(tmp_path_factory) -> Path:
return tmp_path_factory.mktemp("docxskill")
@pytest.fixture(scope="module")
def created(workdir: Path) -> Path:
"""Create a document exercising every create feature."""
png = workdir / "pic.png"
make_png(png)
spec = {
"page": {"width_mm": 210, "height_mm": 297,
"margins_mm": {"top": 25, "bottom": 25,
"left": 20, "right": 20}},
"header": "Report header",
"footer": "Page footer",
"styles": [{"name": "FancyNote", "base": "Normal", "font": "Arial",
"size_pt": 11, "italic": True, "color": "1F4E79"}],
"blocks": [
{"type": "heading", "text": "Main Title", "level": 1},
{"type": "heading", "text": "Section One", "level": 2},
{"type": "paragraph", "runs": [
{"text": "plain "},
{"text": "boldbit", "bold": True},
{"text": " italicbit", "italic": True},
{"text": " underbit", "underline": True}]},
{"type": "paragraph", "text": "Styled note.",
"style": "FancyNote"},
{"type": "bullet_list", "items": ["alpha", "beta"]},
{"type": "numbered_list", "items": ["first", "second"]},
{"type": "table", "header": ["Name", "Qty"],
"rows": [["Widget", "3"], ["Gadget", "5"]],
"style": "Table Grid", "header_bold": True},
{"type": "image", "path": str(png), "width_mm": 30},
{"type": "page_break"},
{"type": "paragraph", "text": "After the break."},
],
}
spec_path = workdir / "spec.json"
spec_path.write_text(json.dumps(spec), encoding="utf-8")
out = workdir / "created.docx"
res = run("docx_create.py", spec_path, out)
assert res["ok"] and out.exists()
return out
class TestCreateAndRead:
def test_text_roundtrip(self, created: Path):
text = run("docx_read.py", created, "--text")
body = "\n".join(text["body"])
for expected in ("Main Title", "plain boldbit italicbit underbit",
"Styled note.", "alpha", "second",
"After the break."):
assert expected in body
assert text["tables"] == [[["Name", "Qty"], ["Widget", "3"],
["Gadget", "5"]]]
assert "Report header" in text["headers"]
assert "Page footer" in text["footers"]
def test_structure(self, created: Path):
st = run("docx_read.py", created, "--structure")
outline = [(h["level"], h["text"]) for h in st["outline"]]
assert (1, "Main Title") in outline
assert (2, "Section One") in outline
assert st["table_count"] == 1
assert st["tables"][0] == {"rows": 3, "cols": 2}
def test_styles_used(self, created: Path):
styles = run("docx_read.py", created, "--styles")["styles"]
for s in ("Heading 1", "FancyNote", "List Bullet", "List Number",
"Table Grid"):
assert s in styles
def test_images_extracted(self, created: Path, workdir: Path):
outdir = workdir / "media"
res = run("docx_read.py", created, "--images", outdir)
assert len(res["images"]) == 1
img = Path(res["images"][0])
assert img.read_bytes().startswith(b"\x89PNG")
def test_run_formatting_persisted(self, created: Path):
doc = Document(str(created))
para = next(p for p in doc.paragraphs if "boldbit" in p.text)
flags = {r.text.strip(): (r.bold, r.italic, r.underline)
for r in para.runs if r.text.strip()}
assert flags["boldbit"][0] is True
assert flags["italicbit"][1] is True
assert flags["underbit"][2] is True
def test_page_setup(self, created: Path):
sec = Document(str(created)).sections[0]
assert round(sec.page_width.mm) == 210
assert round(sec.top_margin.mm) == 25
def test_revisions_detection(self, created: Path):
rev = run("docx_read.py", created, "--revisions")
assert rev["has_tracked_changes"] is False
assert rev["comments"] is False
class TestEdit:
def test_replace_preserves_formatting(self, created: Path, workdir: Path):
out = workdir / "edited.docx"
res = run("docx_edit.py", "replace", created, "--find", "boldbit",
"--replace", "REPLACED", "-o", out)
assert res["replacements"] == 1
doc = Document(str(out))
para = next(p for p in doc.paragraphs if "REPLACED" in p.text)
run_ = next(r for r in para.runs if "REPLACED" in r.text)
assert run_.bold is True # formatting survived
def test_set_cell(self, created: Path, workdir: Path):
out = workdir / "cell.docx"
run("docx_edit.py", "set-cell", created, "--table", "0", "--row",
"1", "--col", "1", "--text", "99", "-o", out)
assert Document(str(out)).tables[0].cell(1, 1).text == "99"
def test_insert_and_delete(self, created: Path, workdir: Path):
out = workdir / "ins.docx"
run("docx_edit.py", "insert", created, "--index", "0", "--text",
"Inserted first", "-o", out)
doc = Document(str(out))
assert doc.paragraphs[0].text == "Inserted first"
out2 = workdir / "del.docx"
run("docx_edit.py", "delete", out, "--index", "0", "-o", out2)
assert Document(str(out2)).paragraphs[0].text != "Inserted first"
def test_apply_style(self, created: Path, workdir: Path):
out = workdir / "styled.docx"
doc = Document(str(created))
idx = next(i for i, p in enumerate(doc.paragraphs)
if p.text == "After the break.")
run("docx_edit.py", "style", created, "--index", str(idx),
"--style", "Heading 2", "-o", out)
doc2 = Document(str(out))
assert doc2.paragraphs[idx].style.name == "Heading 2"
class TestTemplate:
def test_fill_everywhere_non_ascii(self, workdir: Path):
# Build a template: tokens in body, split runs, table, header, footer.
tpl = workdir / "tpl.docx"
doc = Document()
doc.sections[0].header.paragraphs[0].text = "H: {{name}}"
doc.sections[0].footer.paragraphs[0].text = "F: {{date}}"
p = doc.add_paragraph()
p.add_run("Dear {{na") # token split across runs
p.add_run("me}}, hello.")
t = doc.add_table(rows=1, cols=2)
t.cell(0, 0).text = "{{name}}"
t.cell(0, 1).text = "{{ date }}" # spaced variant
doc.add_paragraph("Unfilled: {{missing}}")
doc.save(str(tpl))
values = workdir / "values.json"
values.write_text(
json.dumps({"name": NON_ASCII, "date": "2026-08-08"},
ensure_ascii=False), encoding="utf-8")
out = workdir / "filled.docx"
res = run("docx_template.py", tpl, values, out)
assert res["ok"] is True
assert res["unfilled_tokens"] == ["missing"]
text = run("docx_read.py", out, "--text")
assert f"Dear {NON_ASCII}, hello." in text["body"]
assert text["tables"][0][0] == [NON_ASCII, "2026-08-08"]
assert f"H: {NON_ASCII}" in text["headers"]
assert "F: 2026-08-08" in text["footers"]
def test_strict_fails_on_unfilled(self, workdir: Path):
tpl = workdir / "tpl2.docx"
doc = Document()
doc.add_paragraph("{{gone}}")
doc.save(str(tpl))
values = workdir / "empty.json"
values.write_text("{}", encoding="utf-8")
env = dict(os.environ, LC_ALL="C", PYTHONIOENCODING="utf-8")
proc = subprocess.run(
[sys.executable, str(SCRIPTS / "docx_template.py"), str(tpl),
str(values), str(workdir / "out2.docx"), "--strict"],
capture_output=True, env=env)
assert proc.returncode == 1
payload = json.loads(proc.stdout.decode("utf-8"))
assert payload["unfilled_tokens"] == ["gone"]
# --------------------------------------------------------------- new parity
W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
def q(tag: str) -> str:
return f"{{{W}}}{tag}"
def run_raw(script: str, *args: str):
env = dict(os.environ, LC_ALL="C", PYTHONIOENCODING="utf-8")
return subprocess.run(
[sys.executable, str(SCRIPTS / script), *map(str, args)],
capture_output=True, env=env)
def _add_ins(para, rev_id: int, text: str, author="Editor"):
from lxml import etree
ins = etree.SubElement(para._p, q("ins"))
ins.set(q("id"), str(rev_id))
ins.set(q("author"), author)
ins.set(q("date"), "2026-01-02T03:04:05Z")
r = etree.SubElement(ins, q("r"))
t = etree.SubElement(r, q("t"))
t.text = text
t.set("{http://www.w3.org/XML/1998/namespace}space", "preserve")
def _add_del(para, rev_id: int, text: str, author="Editor"):
from lxml import etree
dele = etree.SubElement(para._p, q("del"))
dele.set(q("id"), str(rev_id))
dele.set(q("author"), author)
dele.set(q("date"), "2026-01-02T03:04:05Z")
r = etree.SubElement(dele, q("r"))
t = etree.SubElement(r, q("delText"))
t.text = text
t.set("{http://www.w3.org/XML/1998/namespace}space", "preserve")
@pytest.fixture()
def tracked(tmp_path: Path) -> Path:
"""Doc with a tracked insertion + deletion in body AND in a table."""
doc = Document()
p = doc.add_paragraph("Base ")
_add_ins(p, 1, "ADDED")
_add_del(p, 2, "REMOVED")
table = doc.add_table(rows=1, cols=1)
cp = table.cell(0, 0).paragraphs[0]
cp.add_run("Cell ")
_add_ins(cp, 3, "CELLADD")
_add_del(cp, 4, "CELLGONE")
path = tmp_path / "tracked.docx"
doc.save(str(path))
return path
class TestRevisions:
def test_list(self, tracked: Path):
res = run("docx_revisions.py", "list", tracked)
revs = {r["id"]: r for r in res["revisions"]}
assert len(revs) == 4
assert revs["1"] == {"id": "1", "author": "Editor",
"date": "2026-01-02T03:04:05Z",
"type": "insertion", "text": "ADDED"}
assert revs["2"]["type"] == "deletion"
assert revs["2"]["text"] == "REMOVED"
assert revs["3"]["text"] == "CELLADD" # inside table
assert revs["4"]["type"] == "deletion"
def test_accept_all(self, tracked: Path, tmp_path: Path):
out = tmp_path / "acc.docx"
res = run("docx_revisions.py", "accept-all", tracked, "-o", out)
assert res["resolved"] == 4
doc = Document(str(out))
assert doc.paragraphs[0].text == "Base ADDED"
assert doc.tables[0].cell(0, 0).text == "Cell CELLADD"
assert run("docx_revisions.py", "list", out)["revisions"] == []
def test_reject_all(self, tracked: Path, tmp_path: Path):
out = tmp_path / "rej.docx"
run("docx_revisions.py", "reject-all", tracked, "-o", out)
doc = Document(str(out))
assert doc.paragraphs[0].text == "Base REMOVED"
assert doc.tables[0].cell(0, 0).text == "Cell CELLGONE"
def test_accept_single_by_id(self, tracked: Path, tmp_path: Path):
out = tmp_path / "one.docx"
res = run("docx_revisions.py", "accept", tracked, "--id", "1",
"-o", out)
assert res["resolved"] == 1
doc = Document(str(out))
assert doc.paragraphs[0].text == "Base ADDED" # del 2 unresolved
remaining = run("docx_revisions.py", "list", out)["revisions"]
assert sorted(r["id"] for r in remaining) == ["2", "3", "4"]
def test_reject_single_by_id(self, tracked: Path, tmp_path: Path):
out = tmp_path / "rone.docx"
run("docx_revisions.py", "reject", tracked, "--id", "2", "-o", out)
doc = Document(str(out))
assert doc.paragraphs[0].text == "Base REMOVED" # ins 1 unresolved
def test_unknown_id_fails(self, tracked: Path, tmp_path: Path):
proc = run_raw("docx_revisions.py", "accept", tracked, "--id",
"999", "-o", tmp_path / "x.docx")
assert proc.returncode == 1
class TestComments:
@pytest.fixture()
def base(self, tmp_path: Path) -> Path:
doc = Document()
doc.add_paragraph("The quarterly revenue rose sharply.")
doc.add_paragraph("Second paragraph.")
path = tmp_path / "base.docx"
doc.save(str(path))
return path
def test_add_list_delete(self, base: Path, tmp_path: Path):
out = tmp_path / "com.docx"
res = run("docx_comments.py", "add", base, "--target",
"quarterly revenue", "--text", "Needs a source",
"--author", "Reviewer", "--initials", "R", "-o", out)
assert res["ok"] is True
cid = res["comment_id"]
listed = run("docx_comments.py", "list", out)["comments"]
assert len(listed) == 1
c = listed[0]
assert c["id"] == cid
assert c["author"] == "Reviewer"
assert c["text"] == "Needs a source"
assert c["anchored_text"] == "quarterly revenue"
assert c["date"]
# document text unchanged by anchoring
text = run("docx_read.py", out, "--text")
assert "The quarterly revenue rose sharply." in text["body"]
out2 = tmp_path / "nocom.docx"
run("docx_comments.py", "delete", out, "--id", cid, "-o", out2)
assert run("docx_comments.py", "list", out2)["comments"] == []
text2 = run("docx_read.py", out2, "--text")
assert "The quarterly revenue rose sharply." in text2["body"]
def test_xml_fallback_path(self, base: Path, tmp_path: Path):
out = tmp_path / "xmlcom.docx"
res = run("docx_comments.py", "add", base, "--target",
"Second paragraph", "--text", "fallback note",
"--author", "Bot", "--xml", "-o", out)
assert res["native_api"] is False
listed = run("docx_comments.py", "list", out)["comments"]
assert listed[0]["text"] == "fallback note"
assert listed[0]["anchored_text"] == "Second paragraph"
# file still opens cleanly
assert Document(str(out)).paragraphs[1].text == "Second paragraph."
def test_missing_target_fails(self, base: Path, tmp_path: Path):
proc = run_raw("docx_comments.py", "add", base, "--target",
"not present", "--text", "x", "-o",
tmp_path / "y.docx")
assert proc.returncode == 1
class TestValidate:
def test_healthy_file_passes(self, created: Path):
res = run("docx_validate.py", created)
assert res["ok"] is True
assert all(i["severity"] != "error" for i in res["issues"])
def test_not_a_zip(self, tmp_path: Path):
bad = tmp_path / "bad.docx"
bad.write_bytes(b"this is not a zip file")
proc = run_raw("docx_validate.py", bad)
assert proc.returncode == 1
rep = json.loads(proc.stdout.decode("utf-8"))
assert rep["issues"][0]["code"] == "not-a-zip"
def test_dangling_rel_and_empty_image(self, created: Path,
tmp_path: Path):
import shutil
import zipfile
broken = tmp_path / "broken.docx"
shutil.copy(created, broken)
# rebuild the zip: drop the image part, zero out nothing else
src = zipfile.ZipFile(str(created))
with zipfile.ZipFile(str(broken), "w") as dst:
for item in src.infolist():
if item.filename.startswith("word/media/"):
dst.writestr(item.filename, b"") # empty image
else:
dst.writestr(item, src.read(item.filename))
proc = run_raw("docx_validate.py", broken)
assert proc.returncode == 1
rep = json.loads(proc.stdout.decode("utf-8"))
codes = {i["code"] for i in rep["issues"]}
assert "empty-image" in codes
def test_missing_style(self, tmp_path: Path):
import zipfile
doc = Document()
doc.add_paragraph("styled", style="Heading 1")
path = tmp_path / "styles.docx"
doc.save(str(path))
# rewrite document.xml to reference a style id that doesn't exist
src = zipfile.ZipFile(str(path))
broken = tmp_path / "badstyle.docx"
with zipfile.ZipFile(str(broken), "w") as dst:
for item in src.infolist():
data = src.read(item.filename)
if item.filename == "word/document.xml":
data = data.replace(b'w:val="Heading1"',
b'w:val="GhostStyle"')
dst.writestr(item, data)
proc = run_raw("docx_validate.py", broken)
assert proc.returncode == 1
rep = json.loads(proc.stdout.decode("utf-8"))
assert any(i["code"] == "missing-style" and "GhostStyle"
in i["detail"] for i in rep["issues"])
class TestNormalize:
def test_merges_split_runs(self, tmp_path: Path):
doc = Document()
p = doc.add_paragraph()
p.add_run("Hel") # identical (no) formatting, split
p.add_run("lo wo")
p.add_run("rld")
b = p.add_run("BOLD1")
b.bold = True
b2 = p.add_run("BOLD2")
b2.bold = True
i = p.add_run("ital")
i.italic = True
path = tmp_path / "split.docx"
doc.save(str(path))
out = tmp_path / "norm.docx"
res = run("docx_edit.py", "normalize", path, "-o", out)
assert res["runs_merged"] == 3 # 2 plain merges + 1 bold merge
doc2 = Document(str(out))
para = doc2.paragraphs[0]
assert para.text == "Hello worldBOLD1BOLD2ital"
assert [r.text for r in para.runs] == \
["Hello world", "BOLD1BOLD2", "ital"]
assert para.runs[1].bold is True
assert para.runs[2].italic is True
class TestFields:
def test_toc_and_page_numbers_via_edit(self, created: Path,
tmp_path: Path):
out = tmp_path / "fields.docx"
run("docx_edit.py", "toc", created, "--index", "0", "-o", out)
run("docx_edit.py", "page-numbers", out)
import zipfile
doc_xml = zipfile.ZipFile(str(out)).read(
"word/document.xml").decode("utf-8")
assert "TOC \\o" in doc_xml
assert "fldChar" in doc_xml
footer_names = [n for n in zipfile.ZipFile(str(out)).namelist()
if n.startswith("word/footer")]
footers = "".join(zipfile.ZipFile(str(out)).read(n).decode("utf-8")
for n in footer_names)
assert "PAGE" in footers and "NUMPAGES" in footers
# still a valid document
assert run("docx_validate.py", out)["ok"] is True
def test_toc_and_footer_in_create_spec(self, tmp_path: Path):
spec = {
"footer_page_numbers": True,
"blocks": [
{"type": "toc"},
{"type": "heading", "text": "Chapter", "level": 1},
],
}
spec_path = tmp_path / "fspec.json"
spec_path.write_text(json.dumps(spec), encoding="utf-8")
out = tmp_path / "fcreate.docx"
run("docx_create.py", spec_path, out)
import zipfile
z = zipfile.ZipFile(str(out))
assert "TOC \\o" in z.read("word/document.xml").decode("utf-8")
footers = "".join(z.read(n).decode("utf-8") for n in z.namelist()
if n.startswith("word/footer"))
assert "NUMPAGES" in footers