diff --git a/CHANGELOG.md b/CHANGELOG.md index 772bbd5bc..5bc51cc17 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,15 @@ follow semantic versioning; release dates are ISO 8601. ### Public API +- **A header or footer sits as far from its page edge as the page puts it.** Nothing was + written, so Word used its own distance — 36pt — and the probe corpus's footer sat 14.5pt + higher than the page draws it, on every page. The engine does not state the distance + either: a zone is a band of a given height against the edge, with its content laid out + inside it from the top, so the distance is read from where the zone's content actually + landed in the resolved layout and written as `w:pgMar/@w:header` or `@w:footer`. Measured + through LibreOffice, the footer now lands within 0.6pt of the page; without a layout the + zone's own padding on that edge stands in. + - **A picture and a list keep the space they hold around themselves.** Neither wrote its own box, so on the probe corpus an image holding 12pt at each edge ran straight into the heading under it, and a four-item checklist came out 13pt short of the page — diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md index 747b27751..995d6ef0b 100644 --- a/docs/recipes/docx-export.md +++ b/docs/recipes/docx-export.md @@ -296,8 +296,14 @@ tint it was flattened to. Recorded, like the other two. Lines, ellipses, standalone shapes, and barcodes are **silently skipped** — they are pure fixed-layout geometry with no semantic equivalent. -Headers/footers, watermarks, and protection options are also ignored by -the current exporter. +The text header and footer slots, watermarks, and protection options are +also ignored by the current exporter. + +A page zone (`session.chrome().zone(...)`) is not: it exports as a real +Word header or footer part, with the page number as a live field, and it +sits as far from its page edge as the page puts it — the distance is read +from where the zone's content landed in the resolved layout and written as +`w:pgMar/@w:header` or `@w:footer`, rather than left to Word's 36pt. The rule of thumb: if the document leans on geometry — shapes, layered designs, precise placement — export PDF for the reader and DOCX only as diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java index a4815e0d0..e2a56dd30 100644 --- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java +++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java @@ -117,6 +117,44 @@ boolean isEmpty() { return fragments.isEmpty(); } + /** + * How far a page zone's content sits from the page edge it belongs to, as laid out on + * the first page. + * + *
Word places a footer by the distance from the page's bottom edge to the bottom of + * the footer, and a header by the distance from the top edge to the top of the header. + * The engine does not state either: a zone is a band of a given height against the + * edge, and its content is laid out inside it from the top. So the distance is read + * from where the content actually landed — the lowest edge of a footer's fragments, the + * highest edge of a header's — rather than rebuilt from the band's parts.
+ * + *Zone fragments are spliced into the graph under {@code @page-zone[page][index]}, + * outside the node paths this index is built from, so they are found by that prefix.
+ * + * @param zoneIndex the zone's position in the session's zone list + * @param header whether it is a header, measured from the top edge + * @param pageHeight the page's height in points + * @return the distance in points, or empty when the layout carries no such zone + */ + OptionalDouble zoneDistanceFromEdge(int zoneIndex, boolean header, double pageHeight) { + String prefix = "@page-zone[0][" + zoneIndex + "]"; + double lowest = Double.POSITIVE_INFINITY; + double highest = Double.NEGATIVE_INFINITY; + for (Map.EntryNothing was written, so Word used its own distance — 36pt — and the probe's footer + * sat 14.5pt higher than the page draws it, on every page. Word holds the distance as + * {@code w:pgMar/@w:header} and {@code @w:footer}, so this is a mapping.
+ * + *The distance is where the zone's content landed in the resolved layout, which is + * the number the page was drawn with. Without a layout it falls back to the zone's own + * padding on that edge — a band's content is laid from its top, so for a footer that is + * the nearer estimate rather than the exact one, and it is only reached when the + * document could not be laid out at all.
+ */ + private void placeZone(XWPFDocument document, DocumentPageZone zone, int index, boolean header) { + CTSectPr sectPr = document.getDocument().getBody().isSetSectPr() + ? document.getDocument().getBody().getSectPr() + : document.getDocument().getBody().addNewSectPr(); + CTPageMar margin = sectPr.isSetPgMar() ? sectPr.getPgMar() : sectPr.addNewPgMar(); + double pageHeight = canvasHeight; + OptionalDouble measured = Double.isNaN(pageHeight) + ? OptionalDouble.empty() + : layout.zoneDistanceFromEdge(index, header, pageHeight); + DocumentInsets padding = zone.getPadding() == null ? DocumentInsets.zero() : zone.getPadding(); + double distance = measured.orElse(header ? padding.top() : padding.bottom()); + if (header) { + margin.setHeader(BigInteger.valueOf(toTwips(distance))); + } else { + margin.setFooter(BigInteger.valueOf(toTwips(distance))); } } @@ -1398,7 +1439,20 @@ private static void addSpacing(XWPFParagraph para, double before, double after) /** Reads a twip measure back, treating an unset one as zero. */ private static long twipsOf(Object measure) { - return measure == null ? 0 : Long.parseLong(String.valueOf(measure)); + Long twips = writtenTwips(measure); + return twips == null ? 0 : twips; + } + + /** + * A twip value this export wrote, read back — or null when it is not a plain number. + * + *XmlBeans hands a measure back as the schema's union, and a measure may legally be a + * string with a unit ({@code 1in}) or a percentage. This export only ever writes plain + * twips, which come back as a number; anything else is not one of its own values and is + * reported as unknown rather than parsed and thrown on.
+ */ + private static Long writtenTwips(Object measure) { + return measure instanceof Number number ? number.longValue() : null; } private static void applyContainerPaint(XWPFParagraph para, ContainerPaint paint) { @@ -2240,9 +2294,8 @@ private static double horizontalMarginsOf(XWPFTableCell cell) { } private static double marginPoints(CTTblWidth margin) { - return margin == null || margin.getW() == null - ? WORD_DEFAULT_CELL_MARGIN_POINTS - : Long.parseLong(String.valueOf(margin.getW())) / POINT_TO_TWIP; + Long twips = margin == null ? null : writtenTwips(margin.getW()); + return twips == null ? WORD_DEFAULT_CELL_MARGIN_POINTS : twips / POINT_TO_TWIP; } /** Word keeps this much clear inside every cell edge unless a table says otherwise. */ @@ -2841,7 +2894,12 @@ private static double usableWidthOf(XWPFTableCell cell, TableGrid.Placement plac double twips = 0; int last = Math.min(placement.column() + placement.colSpan(), grid.sizeOfGridColArray()); for (int index = placement.column(); index < last; index++) { - twips += Long.parseLong(String.valueOf(grid.getGridColArray(index).getW())); + Long column = writtenTwips(grid.getGridColArray(index).getW()); + if (column == null) { + // A column this export did not write as plain twips has no width to add up. + return Double.NaN; + } + twips += column; } double points = twips / POINT_TO_TWIP - horizontalMarginsOf(cell); return points > 0 ? points : Double.NaN; diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java new file mode 100644 index 000000000..f8f4c75d1 --- /dev/null +++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxPageZonePositionTest.java @@ -0,0 +1,146 @@ +package com.demcha.compose.document.backend.semantic.docx; + +import com.demcha.compose.GraphCompose; +import com.demcha.compose.document.api.DocumentSession; +import com.demcha.compose.document.backend.semantic.SemanticBackend; +import com.demcha.compose.document.backend.semantic.SemanticExportContext; +import com.demcha.compose.document.dsl.ParagraphBuilder; +import com.demcha.compose.document.layout.DocumentGraph; +import com.demcha.compose.document.layout.LayoutCanvas; +import com.demcha.compose.document.output.DocumentHeaderFooterZone; +import com.demcha.compose.document.output.DocumentOutputOptions; +import com.demcha.compose.document.output.DocumentPageZone; +import com.demcha.compose.document.style.DocumentInsets; +import org.apache.poi.xwpf.usermodel.XWPFDocument; +import org.junit.jupiter.api.Test; +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPageMar; + +import java.io.ByteArrayInputStream; + +import static org.assertj.core.api.Assertions.assertThat; + +/** + * A header or footer sits as far from its page edge as the page puts it. + * + *Nothing was written, so Word used its own distance — 36pt — and the probe's footer sat + * 14.5pt higher than the page draws it, on every page. The engine does not state the + * distance either: a zone is a band of a given height against the edge, with its content + * laid out inside it from the top. So the distance is read from where the content landed + * in the resolved layout.
+ * + * @author Artem Demchyshyn + */ +class DocxPageZonePositionTest { + + private static final double TWIPS_PER_POINT = 20.0; + + @Test + void aFootersDistanceFollowsItsBandRatherThanWordsDefault() throws Exception { + // The one relation that holds whatever the font measures: content is laid from the + // band's top, so a band 30pt taller lifts its content 30pt further from the edge. + long shallow = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 30, DocumentInsets.zero())); + long deep = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 60, DocumentInsets.zero())); + + assertThat(deep - shallow).isEqualTo(Math.round(30 * TWIPS_PER_POINT)); + assertThat(shallow) + .as("inside the band, and not Word's 720-twip default") + .isBetween(0L, Math.round(30 * TWIPS_PER_POINT)) + .isNotEqualTo(720L); + } + + @Test + void aHeadersDistanceIsItsContentsTopFromThePageTop() throws Exception { + // A header's content starts at the band's top, so its padding is exactly the gap + // between the page's top edge and the content. + long distance = headerDistance(zone(DocumentHeaderFooterZone.HEADER, 40, new DocumentInsets(9, 0, 0, 0))); + + assertThat(distance).isEqualTo(Math.round(9 * TWIPS_PER_POINT)); + } + + @Test + void withoutALayoutAFooterFallsBackToItsOwnPadding() throws Exception { + // A document that cannot be laid out still exports, and a zone's own padding on + // that edge is the nearest thing to the distance it states. + long distance = footerDistanceWithoutLayout( + zone(DocumentHeaderFooterZone.FOOTER, 30, new DocumentInsets(0, 0, 7, 0))); + + assertThat(distance).isEqualTo(Math.round(7 * TWIPS_PER_POINT)); + } + + private static DocumentPageZone zone(DocumentHeaderFooterZone kind, double height, DocumentInsets padding) { + return DocumentPageZone.builder() + .zone(kind) + .height(height) + .padding(padding) + .content(page -> new ParagraphBuilder().name("ZoneLine").text("Chrome").build()) + .build(); + } + + private static long footerDistance(DocumentPageZone zone) throws Exception { + CTPageMar margin = marginOf(exportWithLayout(zone)); + return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter())); + } + + private static long headerDistance(DocumentPageZone zone) throws Exception { + CTPageMar margin = marginOf(exportWithLayout(zone)); + return margin.getHeader() == null ? -1 : Long.parseLong(String.valueOf(margin.getHeader())); + } + + private static long footerDistanceWithoutLayout(DocumentPageZone zone) throws Exception { + CTPageMar margin = marginOf(exportWithoutLayout(zone)); + return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter())); + } + + private static CTPageMar marginOf(byte[] docx) throws Exception { + try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) { + return (CTPageMar) document.getDocument().getBody().getSectPr().getPgMar().copy(); + } + } + + private static byte[] exportWithLayout(DocumentPageZone zone) throws Exception { + try (DocumentSession session = session(zone)) { + return session.export(new DocxSemanticBackend()); + } + } + + /** The same document handed to the backend with no layout, the way a bare caller does. */ + private static byte[] exportWithoutLayout(DocumentPageZone zone) throws Exception { + Captured captured = new Captured(); + try (DocumentSession session = session(zone)) { + session.export(captured); + return new DocxSemanticBackend().export(captured.graph, + new SemanticExportContext(captured.canvas, java.util.List.of(), null, + captured.options)); + } + } + + private static DocumentSession session(DocumentPageZone zone) { + DocumentSession session = GraphCompose.document() + .pageSize(400, 600) + .margin(DocumentInsets.of(40)) + .create(); + session.chrome().zone(zone); + session.pageFlow(page -> page.addParagraph(p -> p.text("Body"))); + return session; + } + + private static final class Captured implements SemanticBackend