From 9cbbf37a9d66ea00f8e57960dcdf0c33f9a7beb4 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Wed, 23 Sep 2026 00:52:16 +0100
Subject: [PATCH 1/2] fix(docx): put a header or footer as far from its page
edge as the page does
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Nothing was written for the distance, so Word used its own 36pt and the
probe corpus's footer sat 14.5pt higher than the page draws it, on every
page.
The engine does not state the distance either: a zone is a band of a
given height against the edge, with its content laid out inside it from
the top. So the distance is read from where the zone's content landed in
the resolved layout — the lowest edge of a footer's fragments, the
highest of a header's — and written as w:pgMar/@w:footer or @w:header.
Without a layout, the zone's own padding on that edge stands in.
Measured through LibreOffice, the footer now lands within 0.6pt of the
page. The recipe also stopped claiming that headers and footers are
ignored: the text slots are, a page zone has not been since it shipped.
---
CHANGELOG.md | 9 ++
docs/recipes/docx-export.md | 10 +-
.../semantic/docx/DocxLayoutMetrics.java | 38 +++++
.../semantic/docx/DocxSemanticBackend.java | 41 ++++-
.../docx/DocxPageZonePositionTest.java | 146 ++++++++++++++++++
5 files changed, 240 insertions(+), 4 deletions(-)
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 772bbd5bc..5bc51cc17 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -8,6 +8,15 @@ follow semantic versioning; release dates are ISO 8601.
### Public API
+- **A header or footer sits as far from its page edge as the page puts it.** Nothing was
+ written, so Word used its own distance — 36pt — and the probe corpus's footer sat 14.5pt
+ higher than the page draws it, on every page. The engine does not state the distance
+ either: a zone is a band of a given height against the edge, with its content laid out
+ inside it from the top, so the distance is read from where the zone's content actually
+ landed in the resolved layout and written as `w:pgMar/@w:header` or `@w:footer`. Measured
+ through LibreOffice, the footer now lands within 0.6pt of the page; without a layout the
+ zone's own padding on that edge stands in.
+
- **A picture and a list keep the space they hold around themselves.** Neither wrote its own
box, so on the probe corpus an image holding 12pt at each edge ran straight into the
heading under it, and a four-item checklist came out 13pt short of the page —
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 747b27751..995d6ef0b 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -296,8 +296,14 @@ tint it was flattened to. Recorded, like the other two.
Lines, ellipses, standalone shapes, and barcodes are **silently skipped**
— they are pure fixed-layout geometry with no semantic equivalent.
-Headers/footers, watermarks, and protection options are also ignored by
-the current exporter.
+The text header and footer slots, watermarks, and protection options are
+also ignored by the current exporter.
+
+A page zone (`session.chrome().zone(...)`) is not: it exports as a real
+Word header or footer part, with the page number as a live field, and it
+sits as far from its page edge as the page puts it — the distance is read
+from where the zone's content landed in the resolved layout and written as
+`w:pgMar/@w:header` or `@w:footer`, rather than left to Word's 36pt.
The rule of thumb: if the document leans on geometry — shapes, layered
designs, precise placement — export PDF for the reader and DOCX only as
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
index a4815e0d0..e2a56dd30 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
@@ -117,6 +117,44 @@ boolean isEmpty() {
return fragments.isEmpty();
}
+ /**
+ * How far a page zone's content sits from the page edge it belongs to, as laid out on
+ * the first page.
+ *
+ * Word places a footer by the distance from the page's bottom edge to the bottom of
+ * the footer, and a header by the distance from the top edge to the top of the header.
+ * The engine does not state either: a zone is a band of a given height against the
+ * edge, and its content is laid out inside it from the top. So the distance is read
+ * from where the content actually landed — the lowest edge of a footer's fragments, the
+ * highest edge of a header's — rather than rebuilt from the band's parts.
+ *
+ * Zone fragments are spliced into the graph under {@code @page-zone[page][index]},
+ * outside the node paths this index is built from, so they are found by that prefix.
+ *
+ * @param zoneIndex the zone's position in the session's zone list
+ * @param header whether it is a header, measured from the top edge
+ * @param pageHeight the page's height in points
+ * @return the distance in points, or empty when the layout carries no such zone
+ */
+ OptionalDouble zoneDistanceFromEdge(int zoneIndex, boolean header, double pageHeight) {
+ String prefix = "@page-zone[0][" + zoneIndex + "]";
+ double lowest = Double.POSITIVE_INFINITY;
+ double highest = Double.NEGATIVE_INFINITY;
+ for (Map.Entry> entry : fragments.entrySet()) {
+ if (!entry.getKey().startsWith(prefix)) {
+ continue;
+ }
+ for (PlacedFragment fragment : entry.getValue()) {
+ lowest = Math.min(lowest, fragment.y());
+ highest = Math.max(highest, fragment.y() + fragment.height());
+ }
+ }
+ if (lowest == Double.POSITIVE_INFINITY) {
+ return OptionalDouble.empty();
+ }
+ return OptionalDouble.of(header ? pageHeight - highest : lowest);
+ }
+
/**
* The path the layout graph addresses a node by, for a note that has to say where in
* the document it came from.
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 2600e2654..3c6fcbb8f 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -117,6 +117,7 @@
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
+import java.util.OptionalDouble;
import java.util.function.Function;
import java.util.concurrent.atomic.AtomicBoolean;
@@ -384,7 +385,8 @@ private void applyPageZones(XWPFDocument document, List zones)
if (policy == null) {
policy = document.createHeaderFooterPolicy();
}
- for (DocumentPageZone zone : zones) {
+ for (int index = 0; index < zones.size(); index++) {
+ DocumentPageZone zone = zones.get(index);
if (zone.getAppliesTo() != null) {
LOG.warn("docx.zone.pagePredicate zone={} — appliesTo cannot be evaluated in a"
+ " semantic export: Word paginates the document, so there is no page to"
@@ -397,10 +399,45 @@ private void applyPageZones(XWPFDocument document, List zones)
if (content == null) {
continue;
}
- XWPFHeaderFooter target = zone.getZone() == DocumentHeaderFooterZone.HEADER
+ boolean header = zone.getZone() == DocumentHeaderFooterZone.HEADER;
+ XWPFHeaderFooter target = header
? policy.createHeader(XWPFHeaderFooterPolicy.DEFAULT)
: policy.createFooter(XWPFHeaderFooterPolicy.DEFAULT);
writeZoneLine(target, content);
+ placeZone(document, zone, index, header);
+ }
+ }
+
+ /**
+ * Puts a header or footer as far from its page edge as the page puts it.
+ *
+ * Nothing was written, so Word used its own distance — 36pt — and the probe's footer
+ * sat 14.5pt higher than the page draws it, on every page. Word holds the distance as
+ * {@code w:pgMar/@w:header} and {@code @w:footer}, so this is a mapping.
+ *
+ * The distance is where the zone's content landed in the resolved layout, which is
+ * the number the page was drawn with. Without a layout it falls back to the zone's own
+ * padding on that edge — a band's content is laid from its top, so for a footer that is
+ * the nearer estimate rather than the exact one, and it is only reached when the
+ * document could not be laid out at all.
+ */
+ private void placeZone(XWPFDocument document, DocumentPageZone zone, int index, boolean header) {
+ CTSectPr sectPr = document.getDocument().getBody().isSetSectPr()
+ ? document.getDocument().getBody().getSectPr()
+ : document.getDocument().getBody().addNewSectPr();
+ CTPageMar margin = sectPr.isSetPgMar() ? sectPr.getPgMar() : sectPr.addNewPgMar();
+ double pageHeight = sectPr.isSetPgSz() && sectPr.getPgSz().getH() != null
+ ? Long.parseLong(String.valueOf(sectPr.getPgSz().getH())) / POINT_TO_TWIP
+ : Double.NaN;
+ OptionalDouble measured = Double.isNaN(pageHeight)
+ ? OptionalDouble.empty()
+ : layout.zoneDistanceFromEdge(index, header, pageHeight);
+ DocumentInsets padding = zone.getPadding() == null ? DocumentInsets.zero() : zone.getPadding();
+ double distance = measured.orElse(header ? padding.top() : padding.bottom());
+ if (header) {
+ margin.setHeader(BigInteger.valueOf(toTwips(distance)));
+ } else {
+ margin.setFooter(BigInteger.valueOf(toTwips(distance)));
}
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java
new file mode 100644
index 000000000..f8f4c75d1
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java
@@ -0,0 +1,146 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.GraphCompose;
+import com.demcha.compose.document.api.DocumentSession;
+import com.demcha.compose.document.backend.semantic.SemanticBackend;
+import com.demcha.compose.document.backend.semantic.SemanticExportContext;
+import com.demcha.compose.document.dsl.ParagraphBuilder;
+import com.demcha.compose.document.layout.DocumentGraph;
+import com.demcha.compose.document.layout.LayoutCanvas;
+import com.demcha.compose.document.output.DocumentHeaderFooterZone;
+import com.demcha.compose.document.output.DocumentOutputOptions;
+import com.demcha.compose.document.output.DocumentPageZone;
+import com.demcha.compose.document.style.DocumentInsets;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.junit.jupiter.api.Test;
+import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPageMar;
+
+import java.io.ByteArrayInputStream;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * A header or footer sits as far from its page edge as the page puts it.
+ *
+ * Nothing was written, so Word used its own distance — 36pt — and the probe's footer sat
+ * 14.5pt higher than the page draws it, on every page. The engine does not state the
+ * distance either: a zone is a band of a given height against the edge, with its content
+ * laid out inside it from the top. So the distance is read from where the content landed
+ * in the resolved layout.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxPageZonePositionTest {
+
+ private static final double TWIPS_PER_POINT = 20.0;
+
+ @Test
+ void aFootersDistanceFollowsItsBandRatherThanWordsDefault() throws Exception {
+ // The one relation that holds whatever the font measures: content is laid from the
+ // band's top, so a band 30pt taller lifts its content 30pt further from the edge.
+ long shallow = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 30, DocumentInsets.zero()));
+ long deep = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 60, DocumentInsets.zero()));
+
+ assertThat(deep - shallow).isEqualTo(Math.round(30 * TWIPS_PER_POINT));
+ assertThat(shallow)
+ .as("inside the band, and not Word's 720-twip default")
+ .isBetween(0L, Math.round(30 * TWIPS_PER_POINT))
+ .isNotEqualTo(720L);
+ }
+
+ @Test
+ void aHeadersDistanceIsItsContentsTopFromThePageTop() throws Exception {
+ // A header's content starts at the band's top, so its padding is exactly the gap
+ // between the page's top edge and the content.
+ long distance = headerDistance(zone(DocumentHeaderFooterZone.HEADER, 40, new DocumentInsets(9, 0, 0, 0)));
+
+ assertThat(distance).isEqualTo(Math.round(9 * TWIPS_PER_POINT));
+ }
+
+ @Test
+ void withoutALayoutAFooterFallsBackToItsOwnPadding() throws Exception {
+ // A document that cannot be laid out still exports, and a zone's own padding on
+ // that edge is the nearest thing to the distance it states.
+ long distance = footerDistanceWithoutLayout(
+ zone(DocumentHeaderFooterZone.FOOTER, 30, new DocumentInsets(0, 0, 7, 0)));
+
+ assertThat(distance).isEqualTo(Math.round(7 * TWIPS_PER_POINT));
+ }
+
+ private static DocumentPageZone zone(DocumentHeaderFooterZone kind, double height, DocumentInsets padding) {
+ return DocumentPageZone.builder()
+ .zone(kind)
+ .height(height)
+ .padding(padding)
+ .content(page -> new ParagraphBuilder().name("ZoneLine").text("Chrome").build())
+ .build();
+ }
+
+ private static long footerDistance(DocumentPageZone zone) throws Exception {
+ CTPageMar margin = marginOf(exportWithLayout(zone));
+ return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter()));
+ }
+
+ private static long headerDistance(DocumentPageZone zone) throws Exception {
+ CTPageMar margin = marginOf(exportWithLayout(zone));
+ return margin.getHeader() == null ? -1 : Long.parseLong(String.valueOf(margin.getHeader()));
+ }
+
+ private static long footerDistanceWithoutLayout(DocumentPageZone zone) throws Exception {
+ CTPageMar margin = marginOf(exportWithoutLayout(zone));
+ return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter()));
+ }
+
+ private static CTPageMar marginOf(byte[] docx) throws Exception {
+ try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) {
+ return (CTPageMar) document.getDocument().getBody().getSectPr().getPgMar().copy();
+ }
+ }
+
+ private static byte[] exportWithLayout(DocumentPageZone zone) throws Exception {
+ try (DocumentSession session = session(zone)) {
+ return session.export(new DocxSemanticBackend());
+ }
+ }
+
+ /** The same document handed to the backend with no layout, the way a bare caller does. */
+ private static byte[] exportWithoutLayout(DocumentPageZone zone) throws Exception {
+ Captured captured = new Captured();
+ try (DocumentSession session = session(zone)) {
+ session.export(captured);
+ return new DocxSemanticBackend().export(captured.graph,
+ new SemanticExportContext(captured.canvas, java.util.List.of(), null,
+ captured.options));
+ }
+ }
+
+ private static DocumentSession session(DocumentPageZone zone) {
+ DocumentSession session = GraphCompose.document()
+ .pageSize(400, 600)
+ .margin(DocumentInsets.of(40))
+ .create();
+ session.chrome().zone(zone);
+ session.pageFlow(page -> page.addParagraph(p -> p.text("Body")));
+ return session;
+ }
+
+ private static final class Captured implements SemanticBackend {
+
+ private DocumentGraph graph;
+ private LayoutCanvas canvas;
+ private DocumentOutputOptions options;
+
+ @Override
+ public String name() {
+ return "capture";
+ }
+
+ @Override
+ public byte[] export(DocumentGraph documentGraph, SemanticExportContext context) {
+ this.graph = documentGraph;
+ this.canvas = context.canvas();
+ this.options = context.outputOptions();
+ return new byte[0];
+ }
+ }
+}
From 58860855b43e73b4efc379fa102b5de536e3235f Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Wed, 23 Sep 2026 01:00:46 +0100
Subject: [PATCH 2/2] fix(docx): read written twips back as numbers instead of
parsing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
placeZone parsed the page height back out of the XML it had just written,
which CodeQL flags as a NumberFormatException nothing catches. It never
needed the XML: the page geometry was written from the canvas, so the
height now comes from the canvas.
The same parse-it-back pattern sat in three older places — the spacing
accumulator, a cell's margin reader, and a nested table's usable width —
each flagged the same way when it landed. XmlBeans hands a measure back
as the schema's union, which may legally be a unit string or a
percentage; this export only ever writes plain twips, and those come back
as a Number. So a written value is read as a Number and anything else is
reported as unknown, with no parsing left in these paths.
---
.../semantic/docx/DocxSemanticBackend.java | 37 +++++++++++++++----
1 file changed, 29 insertions(+), 8 deletions(-)
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 3c6fcbb8f..99844e146 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -180,6 +180,9 @@ public final class DocxSemanticBackend implements SemanticBackend {
/** Space the last body paragraph holds below itself, not yet written — see {@link #owePendingSpacingAfter}. */
private double pendingSpacingAfter;
+ /** The page's height in points, or {@code NaN} when the export has no canvas. */
+ private double canvasHeight = Double.NaN;
+
/** The gap the list being written puts between its items. */
private double pendingItemSpacing;
@@ -308,6 +311,7 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
currentCell = null;
currentCellWidth = Double.NaN;
contentWidth = context.canvas() == null ? Double.MAX_VALUE : context.canvas().innerWidth();
+ canvasHeight = context.canvas() == null ? Double.NaN : context.canvas().height();
try (XWPFDocument document = new XWPFDocument()) {
applyPageGeometry(document, context.canvas());
writeStylesPart(document);
@@ -378,6 +382,8 @@ private void applyOutputOptions(XWPFDocument document, DocumentOutputOptions opt
* absence, and the export says on the log what it could not honor.
*/
private void applyPageZones(XWPFDocument document, List zones) {
+ // The page height the zones are measured against is the canvas's, which is what the
+ // page geometry was written from — not a value parsed back out of the XML.
if (zones == null || zones.isEmpty()) {
return;
}
@@ -426,9 +432,7 @@ private void placeZone(XWPFDocument document, DocumentPageZone zone, int index,
? document.getDocument().getBody().getSectPr()
: document.getDocument().getBody().addNewSectPr();
CTPageMar margin = sectPr.isSetPgMar() ? sectPr.getPgMar() : sectPr.addNewPgMar();
- double pageHeight = sectPr.isSetPgSz() && sectPr.getPgSz().getH() != null
- ? Long.parseLong(String.valueOf(sectPr.getPgSz().getH())) / POINT_TO_TWIP
- : Double.NaN;
+ double pageHeight = canvasHeight;
OptionalDouble measured = Double.isNaN(pageHeight)
? OptionalDouble.empty()
: layout.zoneDistanceFromEdge(index, header, pageHeight);
@@ -1435,7 +1439,20 @@ private static void addSpacing(XWPFParagraph para, double before, double after)
/** Reads a twip measure back, treating an unset one as zero. */
private static long twipsOf(Object measure) {
- return measure == null ? 0 : Long.parseLong(String.valueOf(measure));
+ Long twips = writtenTwips(measure);
+ return twips == null ? 0 : twips;
+ }
+
+ /**
+ * A twip value this export wrote, read back — or null when it is not a plain number.
+ *
+ * XmlBeans hands a measure back as the schema's union, and a measure may legally be a
+ * string with a unit ({@code 1in}) or a percentage. This export only ever writes plain
+ * twips, which come back as a number; anything else is not one of its own values and is
+ * reported as unknown rather than parsed and thrown on.
+ */
+ private static Long writtenTwips(Object measure) {
+ return measure instanceof Number number ? number.longValue() : null;
}
private static void applyContainerPaint(XWPFParagraph para, ContainerPaint paint) {
@@ -2277,9 +2294,8 @@ private static double horizontalMarginsOf(XWPFTableCell cell) {
}
private static double marginPoints(CTTblWidth margin) {
- return margin == null || margin.getW() == null
- ? WORD_DEFAULT_CELL_MARGIN_POINTS
- : Long.parseLong(String.valueOf(margin.getW())) / POINT_TO_TWIP;
+ Long twips = margin == null ? null : writtenTwips(margin.getW());
+ return twips == null ? WORD_DEFAULT_CELL_MARGIN_POINTS : twips / POINT_TO_TWIP;
}
/** Word keeps this much clear inside every cell edge unless a table says otherwise. */
@@ -2878,7 +2894,12 @@ private static double usableWidthOf(XWPFTableCell cell, TableGrid.Placement plac
double twips = 0;
int last = Math.min(placement.column() + placement.colSpan(), grid.sizeOfGridColArray());
for (int index = placement.column(); index < last; index++) {
- twips += Long.parseLong(String.valueOf(grid.getGridColArray(index).getW()));
+ Long column = writtenTwips(grid.getGridColArray(index).getW());
+ if (column == null) {
+ // A column this export did not write as plain twips has no width to add up.
+ return Double.NaN;
+ }
+ twips += column;
}
double points = twips / POINT_TO_TWIP - horizontalMarginsOf(cell);
return points > 0 ? points : Double.NaN;