texts = exportTexts(flow -> flow
.addList("**bold** lead stays intact"));
- assertThat(texts).contains("**bold** lead stays intact");
+ // The session reads markdown, and the item is written as the page sets it: its bold
+ // lead's marks dropped, not one of them taken off as a typed marker.
+ assertThat(texts).contains("bold lead stays intact");
}
@Test
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
index c7d1c2cdd..0fcdef74d 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
@@ -19,11 +19,11 @@
import static org.assertj.core.api.Assertions.assertThat;
/**
- * A paragraph the page reads as markdown is written as the page sets it, and not named
- * ({@link DocxSessionMarkdownTest}); where the page's lines do not tell how it sets it — with no
- * layout — it is written as authored and named. A list item the page reads as markdown is named in
- * the report: the page sets the text its marks style and drops the marks, and the Word file holds
- * the text as authored, marks and all — on the list's note.
+ * A paragraph or a list item the page reads as markdown is written as the page sets it, and not
+ * named ({@link DocxSessionMarkdownTest}, {@link DocxListMarkdownTest}); where the page's lines do
+ * not tell how it sets it — with no layout, or a list composed in a table cell — it is written as
+ * authored and named: the page sets the text its marks style and drops the marks, and the Word file
+ * holds the text as authored, marks and all — a list's items on the list's note.
*
* A session reads markdown unless it is told not to ({@code markdown(false)}), in a paragraph
* or a list item of plain text holding a mark of emphasis or code. Text the page sets as authored —
@@ -34,9 +34,6 @@ class DocxMarkdownReportTest {
private static final String UNMEASURED = "markdown marks are written as letters — whether the page reads them is "
+ "not measured";
- private static final String ITEMS = "its items' markdown marks are written as letters, where the page sets the "
- + "text they mark and drops them";
-
@Test
void aParagraphThePageReadsAsMarkdownIsWrittenSoAndNotNamed() throws Exception {
assertThat(paragraphNotes(true, page -> page.addParagraph("Some **bold** and `code` text"))).isEmpty();
@@ -79,15 +76,15 @@ void aParagraphThePageSetsAsAuthoredIsNotNamed() throws Exception {
}
@Test
- void aListsItemsThePageReadsAsMarkdownAreNamed() throws Exception {
+ void aListsItemsThePageReadsAsMarkdownAreWrittenSoAndNotNamed() throws Exception {
assertThat(listNotes(true, page -> page.addList(list -> list.name("Skills").items("**Java** lead", "Kotlin"))))
- .containsExactly("written as a Word list; " + ITEMS);
+ .isEmpty();
assertThat(listNotes(true, page -> page.addList(list -> list.name("Skills").hangingIndent(true)
.items("**Java** lead", "Kotlin"))))
- .as("markers in a column of their own").containsExactly("written as a Word list; " + ITEMS);
+ .as("markers in a column of their own").isEmpty();
assertThat(listNotes(true, page -> page.addList(list -> list.name("Skills")
.addItem("Languages", child -> child.addItem("**Java**").addItem("Kotlin")))))
- .as("a tree of items").singleElement().asString().endsWith("; " + ITEMS);
+ .as("a tree of items").noneMatch(note -> note.contains("markdown"));
// As the page lays an item out: a marker typed before it is taken off, and none is lost.
assertThat(listNotes(false, page -> page.addList(list -> list.name("Skills").items("* Java", "* Kotlin"))))
.as("a marker typed before an item, markdown off").isEmpty();
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
index 4942a3002..6ccbc4a5c 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
@@ -134,6 +134,36 @@ void thePiecesAreThePagesWhereItsLinesHoldThemSoAndNotOtherwise() {
.as("anything but text").isFalse();
}
+ @Test
+ void piecesAreSplitAtTheEndOfTheLeadTheyOpenWith() {
+ // A nested item's indent and marker, laid out in its text: no-break spaces are letters.
+ String indent = Character.toString(0x00A0).repeat(2);
+ List pieces = DocxMarkdown.read(indent + "◦ **Java** lead", BODY);
+ DocxMarkdown.Split split = DocxMarkdown.split(pieces, indent + "◦ ");
+ assertThat(split.lead()).isEqualTo(BODY);
+ assertThat(split.after()).containsExactly(piece("Java", DocumentTextDecoration.BOLD, 10),
+ piece(" lead", DocumentTextDecoration.DEFAULT, 10));
+ // The lead may end inside a piece and span pieces of one style.
+ assertThat(DocxMarkdown.split(List.of(piece("- ", DocumentTextDecoration.DEFAULT, 10),
+ piece("a b", DocumentTextDecoration.DEFAULT, 10)), "- a").after())
+ .containsExactly(piece(" b", DocumentTextDecoration.DEFAULT, 10));
+ // No lead leaves the pieces whole.
+ assertThat(DocxMarkdown.split(pieces, "")).isEqualTo(new DocxMarkdown.Split(null, pieces));
+
+ assertThat(DocxMarkdown.split(DocxMarkdown.read("*a* **Java**", BODY), "*a* "))
+ .as("a lead the parser reads, its marks dropped").isNull();
+ assertThat(DocxMarkdown.split(List.of(piece("-", DocumentTextDecoration.DEFAULT, 10),
+ piece(" x", DocumentTextDecoration.BOLD, 10), piece(" y", DocumentTextDecoration.DEFAULT, 10)), "- x"))
+ .as("a lead in two styles").isNull();
+ assertThat(DocxMarkdown.split(pieces, indent + "▪ ")).as("another lead").isNull();
+ assertThat(DocxMarkdown.split(List.of(piece("◦ ab", DocumentTextDecoration.DEFAULT, 10),
+ piece("c", DocumentTextDecoration.BOLD, 10)), "▪ ")).as("another lead, in a longer piece").isNull();
+ assertThat(DocxMarkdown.split(List.of(piece("◦ ", DocumentTextDecoration.DEFAULT, 10)), "◦ "))
+ .as("nothing after the lead").isNull();
+ assertThat(DocxMarkdown.split(List.of(piece("◦", DocumentTextDecoration.DEFAULT, 10)), "◦ "))
+ .as("pieces shorter than the lead").isNull();
+ }
+
@Test
void whatThePageMayReadAsMarkdownHoldsAMarkOfEmphasisOrCode() {
assertThat(DocxMarkdown.holdsAMark("a *b*")).isTrue();
From 5fd4fc676973466655dc1aaf4dabfbbbdd54410a Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Wed, 7 Oct 2026 18:26:26 +0100
Subject: [PATCH 2/3] fix(docx): name a list item Word draws the marker of in
another face, and read items for the font table as the page does
A Word list built as a tree of items, with no hangingIndent, has its
items' markers read with them; the parser sets the marker of a bold list
regular, and Word draws it in the list's face, so such an item is written
as authored. It was named only where the page dropped a mark from it:
`Kot_lin` in a bold list, written bold where the page sets it regular,
went unnamed. It is now named whatever marks the page keeps.
The font table read an item's raw label. It now reads each item as the
page lays it out (DocxMarkdown.items): a tree's marker with its item, so
a bold list ships the regular face its marker is written in, and a typed
marker taken off, so `* Java` in a bold list ships no face it does not
use. The item readings move into DocxMarkdown, with a unit test, and
writeItemText is told whether Word draws the marker.
---
CHANGELOG.md | 25 ++--
.../architecture/backend-capability-matrix.md | 2 +-
docs/recipes/docx-export.md | 27 ++--
render-docx/README.md | 8 +-
.../backend/semantic/docx/DocxFontTable.java | 27 +---
.../backend/semantic/docx/DocxMarkdown.java | 80 ++++++++++
.../semantic/docx/DocxSemanticBackend.java | 140 ++++++++----------
.../semantic/docx/DocxListMarkdownTest.java | 68 +++++++--
.../semantic/docx/DocxListParityTest.java | 21 ++-
.../semantic/docx/DocxMarkdownTest.java | 26 ++++
10 files changed, 277 insertions(+), 147 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index e56138756..28966b43c 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -16,27 +16,30 @@ follow semantic versioning; release dates are ISO 8601.
and written in the pieces the page sets it in where those lines hold them, as a paragraph is:
- a flat list's item, after the marker the page sets before its first line;
- an item whose marker stands in a column of its own (`hangingIndent`);
- - a nested item of a list with no `hangingIndent`, which the page lays out after its indent
- and its marker and reads with them: the item's own text is written after them.
+ - an item, at any level, of a list built as a tree of items (`addItem(label, children)`) with
+ no `hangingIndent`, which the page lays out after its indent and its marker and reads with
+ them: the item's own text is written after them.
- **The marker stays where it was.** Word draws a Word list's. A list written as a paragraph per
item writes its marker and nesting indent as characters before the pieces, in the face the
page sets them in.
- **A heading written taller than its item's line is named**, as a paragraph's is.
- - **The font table ships the faces an item's pieces are set in.**
+ - **The font table ships the faces the page sets an item's pieces in**, read off the text the
+ page reads: a tree's marker with its item, a marker typed before an item taken off.
- **Still written as authored, and named:**
- - the items of a list not matched one by one to the layout's: composed in a table cell, an
- item run onto the next page, or a `hangingIndent` list with a blank item the page draws as a
- marker alone;
- - a nested item of a Word list whose marker the page sets in another face than the list's —
- the parser sets a bold list's marker regular — since Word draws it in the list's;
- - a nested item whose marker the parser reads as markdown with it, as `*a*`;
+ - the items of a list not matched one by one to the layout's: with no layout, composed in a
+ table cell, an item run onto the next page, or a `hangingIndent` list with a blank item the
+ page draws as a marker alone;
+ - an item of a Word list built as a tree of items whose marker the page reads with it and sets
+ in another face than the list's — the parser sets a bold list's marker regular — since Word
+ draws it in the list's; it is named whatever marks the page keeps of it;
+ - an item of a tree whose marker the parser reads as markdown with it, as `*a*`;
- one the page sets in other letters than its text, as Arabic;
- an item the parser reads into nothing, as `***`: the page sets none of its text, and the
note now says so, where it named the item's marks.
Across the DOCX fidelity corpus no list item is read as markdown, and the 62 documents are
- byte-identical. `DocxMarkdown.split` takes a nested item's indent and marker off its pieces, with
- a unit test.
+ byte-identical. `DocxMarkdown` reads each item as the page lays it out, and `DocxMarkdown.split`
+ takes a tree's indent and marker off an item's pieces, with unit tests.
- **A DOCX export writes a paragraph the page reads as markdown as the page sets it.** A session
reads markdown unless it is told not to (`markdown(false)`). The page then sets a paragraph of
diff --git a/docs/architecture/backend-capability-matrix.md b/docs/architecture/backend-capability-matrix.md
index 6aba28195..0941d4f46 100644
--- a/docs/architecture/backend-capability-matrix.md
+++ b/docs/architecture/backend-capability-matrix.md
@@ -64,7 +64,7 @@ Payload records live in `core` under
| Capability (payload) | PDF (fixed) | PPTX (fixed) | DOCX (semantic) |
|---|---|---|---|
| Paragraph — pre-wrapped lines, runs, alignment (`ParagraphFragmentPayload`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` (one absolute, wrap-disabled frame per measured line) | ⚠️ semantic paragraphs (`DocxSemanticBackend`) — each run keeps its own style, falling back to the paragraph's when it has none; a centred or right-aligned left-to-right line of its own, of text alone and untracked, that Word sets a point or more wider or narrower at its half-point size has its letters spaced by the difference (`w:spacing`) and its room reckoned from the page's width; a `linkTarget` becomes a `w:hyperlink`, with a relationship for an address or `w:anchor` for one of the document's own anchors, and a run's own link wins over the paragraph's; a paragraph seated off its baseline (`TextVerticalAlign`) has its runs raised or lowered in the line (`w:position`) by the PDF backend's own correction (`ParagraphSeating`), one shift for the paragraph where the page seats each line by its own; Word and LibreOffice stand an exact line's baseline four fifths of the way down it whatever the face, where the page sets it the face's ascent down, so a paragraph whose face puts the two half a point or more apart — Spectral's, not Lato's — has its text moved to the page's baseline in the same position, matched at its middle line (not yet a list item's or a table text cell's; a picture among it moves with it in Word and stays on its own baseline in LibreOffice); lines a container stacks over one another tighter than their face each end halfway between their letters and the next line's (Word draws an exact line's text on screen only inside the line; its PDF export does not cut it), and the last layer of a shape container on one page, where its line runs past the foot, ends at the foot or below its letters; letters two lines share are split halfway so the page does not move, and a stack that holds a picture keeps its lines' own heights; a `bulletOffset` of spaces becomes the paragraph's indent (`w:ind` left, hanging or first line, by `indentStrategy`) in the flow and in cells, not yet over the flow, in an overlay's left-and-right pair, as a badge's initials or in a header or footer; one with letters in it is not written, its wrapped lines still set after the spaces that cover it; an auto-sized paragraph's text is written at its style's size, not the one the page fits it to; a paragraph a session reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, read through the page's own parser, one run a piece in the face, family, colour, tracking and size the page's laid-out lines hold (an auto-sized one's at its style's size), its marks dropped, wherever the lines hold the pieces' letters so; where they are not read, or hold other letters or none (text the parser reads into nothing, which the page sets as nothing), as authored, its marks as letters; a `bookmark(...)` is Word's `HeadingN`, which Word's outline lists by the text of its Word paragraph — an overlay's pair's whole line, one level for both sides — at no level past the ninth. Outside a header or footer, the paragraph's report note (`ParagraphNode`) names each of these where it moves or renames something: the prefix's letters, and the room a path that writes no prefix leaves out where it moves a line; the size written and the size the page fits the text to, to Word's half point; the marks of a paragraph the page read as markdown and the file holds as letters, where its laid-out lines hold fewer of them than its text, not measured where its lines are not read; a markdown heading written taller than the line the page sets it in, which Word cuts on screen; an outline title that is not the text Word lists, a level past the ninth that shares it with another, and the right side's entry where the left holds the line's level |
-| List hanging indent — a marker column and a content column (`ListBuilder.hangingIndent(true)`, `markerGap(...)`) | ✅ marker and content emitted as separate `ParagraphFragmentPayload` fragments at the resolved `markerX` / `contentX` | ✅ the same fragments — the fixed-layout pipeline resolves the geometry before either backend sees it | ⚠️ the top level only. `DocxSemanticBackend` exports a list as a real Word list — `numbering.xml`, `w:numPr` per item, the level carrying the marker — or, with rich items or a drawn marker, as paragraphs; content and nesting are unaffected. With the flag, the top level's marker column is the layout's — the marker's width and `markerGap`, the text and its wrapped lines where the page sets them — where the gap covers what Word may set the marker wider: a picture at its written size, its edges included, or text in the page's face (embedded, or a standard one Word sets in the same widths) grown to its half-point size, half a point clear. A Word list's level then indents and hangs by that column; a list of paragraphs writes the marker, a tab to a stop there, and hangs the item there. Word places content at absolute indents and has no relative-advance primitive, so without the layout's measure the gap could not be honoured; a Word list without the flag that the layout placed and that does not nest takes the page's column too, the spaces the page sets its wrapped lines after, its marker followed by a space (`w:suff`) and an item that wraps measured at Word's half-point size; a list that nests items, a list built as a tree of items (laid out flattened), and a marker the gap does not clear keep the stated column (180 twips, plus 120 per nesting level) — except, in a list of paragraphs, a nested rich item with no marker, which stands where the layout set its text, its measure weighed at Word's half-point sizes, where the layout's items are matched to the list's; a list that nests only such items sets its top level at the page's column too. The report counts, on the list, the items that stand at a stated column, a space past their marker or two spaces a level in, and names a centred or right-aligned list written flush left, a lineSpacing not written where the layout's items are not the list's own and one wraps (in a list composed in a table cell, its wrapping not measured), a continuationIndent not written where an item of a markerless list or a tree of items without the flag wraps or its wrapping is not measured, the rows the page draws as a marker alone for blank items of a flagged list, which are not written, and what items the page reads as markdown lose. An item of plain text the page reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, matched to the lines the page laid it out in and read as the page lays it out (a flat item after the marker the page sets before it; a nested item of a list without the flag after the indent and marker the page reads with it), one run a piece in the face and size those lines hold, Word or the file drawing the marker where it did; it is written as authored, its marks as letters, and named where its list's items are not matched one by one to the layout's (composed in a table cell, an item run onto the next page, a flagged list with a blank item), where its lines hold other letters, where a Word list's nested marker is set in another face on the page than the list's, and where the parser reads it into nothing; a markdown heading written taller than its item's line is named |
+| List hanging indent — a marker column and a content column (`ListBuilder.hangingIndent(true)`, `markerGap(...)`) | ✅ marker and content emitted as separate `ParagraphFragmentPayload` fragments at the resolved `markerX` / `contentX` | ✅ the same fragments — the fixed-layout pipeline resolves the geometry before either backend sees it | ⚠️ the top level only. `DocxSemanticBackend` exports a list as a real Word list — `numbering.xml`, `w:numPr` per item, the level carrying the marker — or, with rich items or a drawn marker, as paragraphs; content and nesting are unaffected. With the flag, the top level's marker column is the layout's — the marker's width and `markerGap`, the text and its wrapped lines where the page sets them — where the gap covers what Word may set the marker wider: a picture at its written size, its edges included, or text in the page's face (embedded, or a standard one Word sets in the same widths) grown to its half-point size, half a point clear. A Word list's level then indents and hangs by that column; a list of paragraphs writes the marker, a tab to a stop there, and hangs the item there. Word places content at absolute indents and has no relative-advance primitive, so without the layout's measure the gap could not be honoured; a Word list without the flag that the layout placed and that does not nest takes the page's column too, the spaces the page sets its wrapped lines after, its marker followed by a space (`w:suff`) and an item that wraps measured at Word's half-point size; a list that nests items, a list built as a tree of items (laid out flattened), and a marker the gap does not clear keep the stated column (180 twips, plus 120 per nesting level) — except, in a list of paragraphs, a nested rich item with no marker, which stands where the layout set its text, its measure weighed at Word's half-point sizes, where the layout's items are matched to the list's; a list that nests only such items sets its top level at the page's column too. The report counts, on the list, the items that stand at a stated column, a space past their marker or two spaces a level in, and names a centred or right-aligned list written flush left, a lineSpacing not written where the layout's items are not the list's own and one wraps (in a list composed in a table cell, its wrapping not measured), a continuationIndent not written where an item of a markerless list or a tree of items without the flag wraps or its wrapping is not measured, the rows the page draws as a marker alone for blank items of a flagged list, which are not written, and what items the page reads as markdown lose. An item of plain text the page reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, matched to the lines the page laid it out in and read as the page lays it out (a flat item after the marker the page sets before it; an item of a list built as a tree of items without the flag after the indent and marker the page reads with it), one run a piece in the face and size those lines hold, Word or the file drawing the marker where it did, the font table shipping the faces the page sets the pieces in; it is written as authored, its marks as letters, and named where its list's items are not matched one by one to the layout's (no layout, composed in a table cell, an item run onto the next page, a flagged list with a blank item), where its lines hold other letters, where the parser reads a tree's marker as markdown with the item (`*a*`), where a Word list's tree item has its marker set in another face on the page than the list's (named whatever marks the page keeps), and where the parser reads it into nothing; a markdown heading written taller than its item's line is named |
| Inline code/badge chips (`InlineBackground` on text spans) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ⚠️ `DocxSemanticBackend` — the fill becomes the run's own `w:shd`, in a paragraph and in a list item alike, so a badge still reads as a badge. What Word has no way to say is the shape: shading covers the glyph box, so the corner radius and the padding above and below the letters are not in the file, and the export records them. The padding beside the letters is written as the room it takes (`spaceAfterTheLastLetter`): character spacing after the chip's last letter, shaded with it, and after the letter before the chip, unshaded. A chip opening its line or following a picture has no letter before it, so its left padding is not in the file; no space is written after right-to-left letters or after a symbol or emoji. The export records, chip by chip, how each side was written. LibreOffice sets no spacing after a line's last letter, so it does not apply the right padding of a chip that ends a line. A `w:shd` fill is opaque, so a translucent chip is flattened first against what Word paints underneath it — the paragraph's shading, the cell's, or else the colour the page paints under the paragraph, a page background included — so the chip agrees with the file it is in and shows the colour the PDF shows. It stops being translucent, and that is recorded with the rest |
| Inline images (`ParagraphImageSpan`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ✅ `DocxSemanticBackend.writeInlinePicture` (a picture in its own run where it sits among the words, at its size, inside the run's or the paragraph's link; raised or lowered by `w:position` to where the page's alignment and `baselineOffset` put it, from the layout's measure of the paragraph's first line — in a list, the list's text on a line as tall as the item's own tallest picture; LibreOffice ignores `w:position` on a picture and stands it on the baseline, so a picture the export draws itself (icon, emoji, shape) that the page raises carries the rise as transparent rows and needs no `w:position`, while one the page lowers stands in LibreOffice higher than on the page by as much as the page lowers it — up to the text's descent for a centred icon as tall as its line; the editor clips a picture to an exact line height, so a paragraph holding a picture that leaves its text — past the ascent or the descent, in Word's placement or on the baseline — has its lines written at least the height the picture reaches, grown by the editor rather than clipped, every line of the paragraph since Word has one line height for it, and each as tall as the editor's font makes it — for 14pt text about 2.5pt taller than the page's in LibreOffice; a picture inside its text in both editors keeps the exact height; a paragraph of one line of text in a Word paragraph of its own, with room above for its pictures' reach, keeps an exact line at the page's height of it, the pictures set in it where the page puts them in Word and what their ink reaches past it taken from the gaps around it, and in LibreOffice a lowered picture there stands higher and loses what passes the line's top; its description is the text it stands for or empty) |
| Inline vector shapes (`ParagraphShapeSpan`) | ✅ `PdfParagraphFragmentRenderHandler` | ⚠️ `PptxParagraphFragmentRenderHandler` + `PptxInlineGeometry` (distinct per-corner radii render with the top-left radius — single-adjust preset) | ⚠️ `DocxSemanticBackend.writeInlinePicture` + `DocxShapePictures` (a transparent PNG drawn by the shared `InlineSvgRasters` from the outline, fill and stroke — every outline kind, each layer centred in the run's box — placed as an inline picture is; the picture takes as far as the stroked ink reaches past the outline — half the stroke on an edge, more at a sharp corner's miter — and a pixel on each side, measured side by side, and is lowered by what it takes below, so no edge is cut and a shape takes that much more room in the line; a list marker that draws a disc is its picture; at the top level of a `hangingIndent(true)` list that does not nest it is followed by a tab to where the layout starts the item's text, the item's lines hanging there, when the picture clears that stop, else by a space) |
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 5980d6295..1eda66158 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -465,18 +465,21 @@ An item of plain text the page reads as markdown is written as the page sets it,
is: matched to the lines the page laid it out in, and written in the pieces the page sets it in
where those lines hold them — its marks dropped, each piece in the page's face and size. A flat
list's item is read after the marker the page sets before its first line; an item whose marker
-stands in a column of its own, as its text alone; a nested item of a list without
-`hangingIndent`, which the page lays out after its indent and its marker and reads with them, as
-its own text after them. The marker stays where it was: Word draws a Word list's, and a list of
-paragraphs writes its marker and nesting indent as characters before the pieces, in the face the
-page sets them in. The report names a heading written taller than its item's line, which Word
-cuts on screen. An item is written as authored, its marks as letters, and named where its list's
-items are not matched one by one to the layout's — composed in a table cell, an item run onto
-the next page, a `hangingIndent` list with a blank item the page draws as a marker alone — where
-its lines hold other letters than its pieces, as Arabic, or where the page sets a nested item's
-marker in another face than the list's in a Word list, whose marker Word draws in the list's. An
-item the parser reads into nothing, as `***`, is written as authored too, and named as an item
-the page sets none of the text of.
+stands in a column of its own, as its text alone; an item, at any level, of a list built as a
+tree of items without `hangingIndent`, which the page lays out after its indent and its marker
+and reads with them, as its own text after them. The marker stays where it was: Word draws a
+Word list's, and a list of paragraphs writes its marker and nesting indent as characters before
+the pieces, in the face the page sets them in. The font table ships the faces the page sets the
+pieces in. The report names a heading written taller than its item's line, which Word cuts on
+screen. An item is written as authored, its marks as letters, and named where its list's items
+are not matched one by one to the layout's — with no layout, composed in a table cell, an item
+run onto the next page, a `hangingIndent` list with a blank item the page draws as a marker
+alone — where its lines hold other letters than its pieces, as Arabic, or where the parser reads
+the marker of a tree's item as markdown with it, as `*a*`. An item of a Word list built as a tree
+whose marker the page sets in another face than the list's — its parser sets a bold list's marker
+regular — is written as authored too, since Word draws the marker in the list's, and named
+whatever marks the page keeps of it. An item the parser reads into nothing, as `***`, is written
+as authored, and named as an item the page sets none of the text of.
## What a panel keeps and loses
diff --git a/render-docx/README.md b/render-docx/README.md
index 016dec9db..f608473b2 100644
--- a/render-docx/README.md
+++ b/render-docx/README.md
@@ -130,9 +130,11 @@ What is not written — each one is named in the export report
- the marks of items the page reads as markdown, where they are written as letters: an item is
written as the page sets it — its marks dropped, each piece in the page's face and size —
wherever its lines show how, and as authored where its list's items are not matched one by one
- to the layout's (a cell, an item run onto the next page, a hanging-indent list with a blank
- item), its lines hold other letters, a Word list's nested marker stands in another face on the
- page, or the page sets none of its text; and a markdown heading written taller than its line.
+ to the layout's (no layout, a cell, an item run onto the next page, a hanging-indent list with a
+ blank item), its lines hold other letters, the parser reads a tree's marker as markdown with it
+ (`*a*`), or the page sets none of its text; a Word list's item whose marker the page sets in
+ another face than the list's, written as authored whatever marks the page keeps; and a
+ markdown heading written taller than its line.
- **What a paragraph's own fields set where Word cannot hold it**, named in the export report on
the paragraph outside a header or footer (a page zone's are named on the zone):
- the size an auto-sized paragraph's text is fitted to, where Word, to its half point, holds it
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
index a493833ab..8e3ab945d 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
@@ -5,7 +5,6 @@
import com.demcha.compose.document.node.InlineHighlightRun;
import com.demcha.compose.document.node.InlineRun;
import com.demcha.compose.document.node.InlineTextRun;
-import com.demcha.compose.document.node.ListItem;
import com.demcha.compose.document.node.ListNode;
import com.demcha.compose.document.node.ParagraphNode;
import com.demcha.compose.document.node.TableNode;
@@ -337,8 +336,13 @@ private static void collectFonts(DocumentNode node, Map> int
} else if (node instanceof ListNode list) {
add(list.textStyle(), into);
if (list.textStyle() != null) {
- list.items().forEach(item -> addPieces(item, list.textStyle(), into));
- addPieces(list.nestedItems(), list.textStyle(), into);
+ // Read as the page reads them: a typed marker taken off, a tree's indent and
+ // marker read with the item, in the face the page sets them in.
+ for (DocxMarkdown.ItemReading item : DocxMarkdown.items(list)) {
+ if (DocxMarkdown.holdsAMark(item.text())) {
+ DocxMarkdown.read(item.text(), list.textStyle()).forEach(piece -> add(piece.style(), into));
+ }
+ }
}
} else if (node instanceof TableNode table) {
add(styleOf(table.defaultCellStyle()), into);
@@ -351,23 +355,6 @@ private static void collectFonts(DocumentNode node, Map> int
}
}
- /** The faces the markdown of a tree of items of plain text sets pieces of them in. */
- private static void addPieces(List items, DocumentTextStyle style, Map> into) {
- for (ListItem item : items) {
- if (!item.isRich()) {
- addPieces(item.label(), style, into);
- }
- addPieces(item.children(), style, into);
- }
- }
-
- /** The faces the markdown of an item's text sets pieces of it in. */
- private static void addPieces(String text, DocumentTextStyle style, Map> into) {
- if (DocxMarkdown.holdsAMark(text)) {
- DocxMarkdown.read(text, style).forEach(piece -> add(piece.style(), into));
- }
- }
-
private static DocumentTextStyle styleOf(DocumentTableCell cell) {
return cell == null ? null : styleOf(cell.style());
}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
index ed08af0a1..7d7642f4d 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
@@ -3,6 +3,9 @@
import com.demcha.compose.document.layout.payloads.ParagraphLine;
import com.demcha.compose.document.layout.payloads.ParagraphSpan;
import com.demcha.compose.document.layout.payloads.ParagraphTextSpan;
+import com.demcha.compose.document.node.ListItem;
+import com.demcha.compose.document.node.ListMarker;
+import com.demcha.compose.document.node.ListNode;
import com.demcha.compose.document.node.ParagraphNode;
import com.demcha.compose.document.style.DocumentLetterSpacing;
import com.demcha.compose.document.style.DocumentTextDecoration;
@@ -63,6 +66,83 @@ static boolean mayRead(ParagraphNode node) {
return (node.inlineRuns() == null || node.inlineRuns().isEmpty()) && holdsAMark(node.text());
}
+ /**
+ * The indent a list with no {@code hangingIndent} sets an item of a tree of items in, a level
+ * at a time: two no-break spaces ({@code TextFlowSupport}, where it flattens the tree into
+ * labels), which the parser reads as letters.
+ */
+ private static final String NESTED_ITEM_INDENT = Character.toString(0x00A0).repeat(2);
+
+ /**
+ * A list item's text as the page lays it out, and reads it where its session reads markdown.
+ *
+ * @param text the text the page lays the item out in and reads
+ * @param lead the characters it opens with that the file writes apart from the item's own
+ * text: the indent and marker a list of a tree of items with no
+ * {@code hangingIndent} lays out in its text; empty for none
+ * @param prefix the prefix the page sets before its first line, apart from its text: a flat
+ * list's marker, where it has no {@code hangingIndent}; empty for none
+ */
+ record ItemReading(String text, String lead, String prefix) {
+ }
+
+ /**
+ * A flat list's item as the page reads it: its text, a marker typed before it taken off, after
+ * the list's marker where the page sets that before its first line.
+ *
+ * @param normalized the item's text, a typed marker taken off
+ */
+ static ItemReading flatItem(ListNode list, String normalized) {
+ return new ItemReading(normalized, "",
+ !list.hangingIndent() && list.marker().isVisible() ? list.marker().prefix() : "");
+ }
+
+ /**
+ * An item of a tree of items as the page reads it: with {@code hangingIndent}, its label, a
+ * marker typed before it taken off; without it, its label after its depth's indent and its
+ * marker, as the page flattens the tree into labels, nothing taken off.
+ *
+ * @param marker the marker the item takes, its own or its depth's
+ * @return its reading, {@code null} for an item of runs, which the page never reads
+ */
+ static ItemReading nestedItem(ListNode list, ListItem item, int depth, ListMarker marker) {
+ if (item.isRich()) {
+ return null;
+ }
+ if (list.hangingIndent()) {
+ return new ItemReading(ListMarker.normalizeItemText(item.label(), list.normalizeMarkers()), "", "");
+ }
+ String lead = NESTED_ITEM_INDENT.repeat(depth) + (marker.isVisible() ? marker.prefix() : "");
+ return new ItemReading(lead + item.label(), lead, "");
+ }
+
+ /**
+ * Every item of plain text a list writes, as the page reads it: its flat items, then its tree
+ * of items depth first, each taking its own marker or its depth's.
+ */
+ static List items(ListNode list) {
+ List readings = new ArrayList<>();
+ for (String item : list.items()) {
+ String normalized = ListMarker.normalizeItemText(item, list.normalizeMarkers());
+ if (!normalized.isBlank()) {
+ readings.add(flatItem(list, normalized));
+ }
+ }
+ addItems(list, list.nestedItems(), 0, readings);
+ return List.copyOf(readings);
+ }
+
+ private static void addItems(ListNode list, List items, int depth, List into) {
+ for (ListItem item : items) {
+ ItemReading reading = nestedItem(list, item,
+ depth, item.marker() != null ? item.marker() : ListMarker.defaultForDepth(depth));
+ if (reading != null) {
+ into.add(reading);
+ }
+ addItems(list, item.children(), depth + 1, into);
+ }
+ }
+
/**
* One piece of text the page sets in one style.
*
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 8049f5654..3fbd97a97 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -3950,7 +3950,6 @@ private void writeList(XWPFDocument document,
: new java.util.ArrayDeque<>();
boolean previousMatched = listItemsMatched;
listItemsMatched = !laidOut.isEmpty() && laidOut.size() == itemCount(list);
- boolean matched = listItemsMatched;
List previousWritten = listItemsWritten;
List written = new ArrayList<>();
listItemsWritten = written;
@@ -3986,7 +3985,7 @@ private void writeList(XWPFDocument document,
owePendingSpacingAfter(list.margin().bottom() + list.padding().bottom());
reportWrittenWithout(list, itemCount(list) == 0 ? "writes no paragraph"
: numId != null ? "written as a Word list" : "written as a paragraph per item",
- listLost(list, laidOut, atTheirColumn, matched, written));
+ listLost(list, laidOut, atTheirColumn, written));
}
/**
@@ -4009,14 +4008,15 @@ private void writeList(XWPFDocument document,
*
* @param laidOut the items as the layout laid them out
* @param atTheirColumn how many of its items stand where the page sets them
- * @param matched whether the layout's items are matched to the list's, one by one
* @param written its items of plain text as written
*/
private List listLost(com.demcha.compose.document.node.ListNode list,
List laidOut, int atTheirColumn,
- boolean matched, List written) {
+ List written) {
List lost = new ArrayList<>();
int items = itemCount(list);
+ // The layout's items matched to the list's one by one, as writeList matches them.
+ boolean matched = !laidOut.isEmpty() && laidOut.size() == items;
int markerRows = markerOnlyRows(list);
if (items > 0) {
lost.addAll(itemsLost(list, laidOut, items, markerRows, atTheirColumn));
@@ -4033,10 +4033,12 @@ private List listLost(com.demcha.compose.document.node.ListNode list,
* What a list's items lose where the page reads them as markdown, each item matched to the
* lines the page laid it out in. An item of plain text holding a mark of emphasis or code is
* written in the pieces the page sets it in where its lines say so ({@link #itemPieces}); a
- * heading among them in a line Word cuts it in is named ({@link #headingCut}). One the page
- * reads into nothing is written as authored and named; one written as authored, its marks
- * and all, is named where its lines, less the prefix the page sets before its first line, hold
- * fewer marks than it ({@link #marksDropped}).
+ * heading among them in a line Word cuts it in is named ({@link #headingCut}). One whose marker
+ * Word draws in the list's face, where the page sets it in another, is written as authored and
+ * named, whatever marks the page keeps of it. One the page reads into nothing is written as
+ * authored and named; any other written as authored, its marks and all, is named where its
+ * lines, less the prefix the page sets before its first line, hold fewer marks than it
+ * ({@link #marksDropped}).
*
* @param items how many items the list writes
* @param written its items of plain text as written
@@ -4052,8 +4054,17 @@ private List itemsMarkdownLost(com.demcha.compose.document.node.ListNode
break;
}
}
+ long refused = written.stream().filter(ItemWritten::refused).count();
+ if (refused > 0) {
+ boolean one = refused == 1;
+ lost.add(refused + " of its " + items + " items " + (one ? "is" : "are")
+ + " written as authored, where the page reads " + (one ? "it" : "them")
+ + " as markdown and sets " + (one ? "its marker" : "their markers")
+ + " in another face than the list's, the face Word draws a list's marker in");
+ }
List asAuthored = written.stream()
- .filter(item -> item.pieces() == null && DocxMarkdown.holdsAMark(item.reading().text())).toList();
+ .filter(item -> item.pieces() == null && !item.refused() && DocxMarkdown.holdsAMark(item.reading().text()))
+ .toList();
List nothing = asAuthored.stream().filter(item -> setsNoneOfIt(item, list.textStyle())).toList();
if (!nothing.isEmpty()) {
boolean one = nothing.size() == 1;
@@ -4104,8 +4115,9 @@ private static long lettersIn(String text) {
/**
* What a list's items lose where the page reads them as markdown, its items not matched to the
- * layout's — composed in a table cell, or one run onto the next page — and so each written as
- * authored, as a paragraph's text is ({@link #markdownLost}): an item of plain text holding a
+ * layout's — with no layout, composed in a table cell, one run onto the next page, or a blank
+ * item of a {@code hangingIndent} list the page draws as a marker alone — and so each written
+ * as authored, as a paragraph's text is ({@link #markdownLost}): an item of plain text holding a
* mark of emphasis or code is set as markdown, its marks dropped. Its text is read as the page
* lays it out, a marker typed before it taken off
* ({@link com.demcha.compose.document.node.ListMarker#normalizeItemText}); its laid-out lines
@@ -4136,73 +4148,30 @@ private List itemsMarkdownUnmatched(com.demcha.compose.document.node.Lis
return marks == null ? List.of() : List.of(marks);
}
- /**
- * A list item's text as the page reads it where its session reads markdown.
- *
- * @param text the text the page lays the item out in and reads
- * @param lead the characters it opens with that the file writes apart from the item's own
- * text: a nested item's indent and marker, which a list with no
- * {@code hangingIndent} lays out in its text; empty for none
- * @param prefix the prefix the page sets before its first line, apart from its text: a flat
- * list's marker, where it has no {@code hangingIndent}; empty for none
- */
- private record ItemReading(String text, String lead, String prefix) {
- }
-
/**
* A list item of plain text as written.
*
* @param reading its text as the page reads it
* @param lines the lines the page laid it out in, empty where its list's items are not matched
* @param pieces the pieces it was written in, {@code null} where it was written as authored
+ * @param refused whether it was written as authored though its lines hold the pieces the page
+ * sets it in: Word draws its marker, which the page sets in another face
*/
- private record ItemWritten(ItemReading reading,
+ private record ItemWritten(DocxMarkdown.ItemReading reading,
List lines,
- List pieces) {
+ List pieces, boolean refused) {
}
/**
- * The indent a list with no {@code hangingIndent} sets a nested item's text in, a level at a
- * time: two no-break spaces, which the markdown parser reads as letters
- * ({@code TextFlowSupport}, where it flattens a tree of items into labels).
- */
- private static final String NESTED_ITEM_INDENT = Character.toString(0x00A0).repeat(2);
-
- /** A flat list's item as the page reads it. */
- private static ItemReading flatItemReading(com.demcha.compose.document.node.ListNode list, String normalized) {
- return new ItemReading(normalized, "",
- !list.hangingIndent() && list.marker().isVisible() ? list.marker().prefix() : "");
- }
-
- /**
- * A nested item as the page reads it: with {@code hangingIndent} its label, a marker typed
- * before it taken off; without it, its label after its depth's indent and its marker, as the
- * page flattens a tree of items into labels, none taken off. {@code null} for an item of runs.
- */
- private static ItemReading nestedItemReading(com.demcha.compose.document.node.ListNode list,
- com.demcha.compose.document.node.ListItem item, int depth,
- com.demcha.compose.document.node.ListMarker marker) {
- if (item.isRich()) {
- return null;
- }
- if (list.hangingIndent()) {
- return new ItemReading(com.demcha.compose.document.node.ListMarker
- .normalizeItemText(item.label(), list.normalizeMarkers()), "", "");
- }
- String lead = NESTED_ITEM_INDENT.repeat(depth) + (marker.isVisible() ? marker.prefix() : "");
- return new ItemReading(lead + item.label(), lead, "");
- }
-
- /**
- * An item's text in the pieces the page sets it in, after the lead the file writes apart, where
- * the page reads it as markdown and the lines it laid the item out in say so
- * ({@link #pagePieces}); {@code null} where it is written as authored: its list's items are
- * not matched to the layout's, or the lead is not set as the page sets the rest of the
- * pieces' first.
+ * An item's text in the pieces the page sets it in, split at the end of the lead the file
+ * writes apart from them, where the page reads it as markdown and the lines it laid the item
+ * out in hold the pieces ({@link #pagePieces}); {@code null} where it is written as authored:
+ * its list's items are not matched to the layout's, its lines do not hold the pieces, or the
+ * pieces do not open with the lead's characters in one style ({@link DocxMarkdown#split}).
*
* @param laidOut how the layout set the item, {@code null} where its list's items are not matched
*/
- private static DocxMarkdown.Split itemPieces(ItemReading reading, DocumentTextStyle style,
+ private static DocxMarkdown.Split itemPieces(DocxMarkdown.ItemReading reading, DocumentTextStyle style,
DocxLayoutMetrics.ItemText laidOut) {
if (reading == null || laidOut == null) {
return null;
@@ -4215,18 +4184,23 @@ private static DocxMarkdown.Split itemPieces(ItemReading reading, DocumentTextSt
* Writes an item's text: as authored in one run, or in the pieces the page sets it in, and
* records it for the list's note.
*
- * @param before what the file writes before the item's own text — its indent and marker,
- * where Word does not draw the marker — in the lead's style where the page
- * sets the lead in the pieces
- * @param label the item's text as authored
- * @param laidOut how the layout set the item, {@code null} where its list's items are not matched
+ * @param before what the file writes before the item's own text — its indent and
+ * marker, where Word does not draw the marker — in the lead's style
+ * where the page sets the lead in the pieces
+ * @param label the item's text as authored
+ * @param wordDrawsMarker whether Word draws the item's marker, a Word list's
+ * @param laidOut how the layout set the item, {@code null} where its list's items are
+ * not matched
*/
private void writeItemText(XWPFParagraph para, DocumentTextStyle style, String before, String label,
- ItemReading reading, DocxLayoutMetrics.ItemText laidOut) {
+ boolean wordDrawsMarker, DocxMarkdown.ItemReading reading,
+ DocxLayoutMetrics.ItemText laidOut) {
DocxMarkdown.Split split = itemPieces(reading, style, laidOut);
- // Word draws a list's marker in the list's style. Where the page sets the lead in another —
- // its parser sets a bold list's nested marker regular — the item is written as authored.
- if (split != null && before.isEmpty() && split.lead() != null && !split.lead().equals(style)) {
+ // Word draws a Word list's marker in the list's style. Where the page sets the marker it
+ // reads with the item in another — its parser sets a bold list's marker regular — the item
+ // is written as authored, and named.
+ boolean refused = split != null && wordDrawsMarker && split.lead() != null && !split.lead().equals(style);
+ if (refused) {
split = null;
}
if (split == null) {
@@ -4247,7 +4221,7 @@ private void writeItemText(XWPFParagraph para, DocumentTextStyle style, String b
}
if (reading != null) {
listItemsWritten.add(new ItemWritten(reading, laidOut == null ? List.of() : laidOut.lines(),
- split == null ? null : split.after()));
+ split == null ? null : split.after(), refused));
}
}
@@ -4434,7 +4408,7 @@ private int writeListItems(XWPFDocument document,
continue;
}
java.util.OptionalDouble lineHeight = layout.lineHeight(list);
- ItemReading reading = flatItemReading(list, normalized);
+ DocxMarkdown.ItemReading reading = DocxMarkdown.flatItem(list, normalized);
boolean atItsColumn;
if (list.marker().isRich()) {
// A drawn marker's pieces are runs, so the row is written the way
@@ -4484,7 +4458,7 @@ private int writeNestedItem(XWPFDocument document,
? item.marker()
: com.demcha.compose.document.node.ListMarker.defaultForDepth(depth);
java.util.OptionalDouble lineHeight = layout.lineHeight(list);
- ItemReading reading = nestedItemReading(list, item, depth, marker);
+ DocxMarkdown.ItemReading reading = DocxMarkdown.nestedItem(list, item, depth, marker);
boolean atItsColumn;
if (item.isRich() || marker.isRich()) {
// A nested item stands after its depth's indent, and an item may carry a marker of
@@ -4528,7 +4502,8 @@ private int writeNestedItem(XWPFDocument document,
* and the nesting indent as characters
*/
private void writeListLine(XWPFDocument document, DocumentTextStyle style,
- String marker, String label, ItemReading reading, int depth, BigInteger numId,
+ String marker, String label, DocxMarkdown.ItemReading reading, int depth,
+ BigInteger numId,
java.util.OptionalDouble lineHeight) {
spaceBeforeTheNextItem();
XWPFParagraph para = newBodyParagraph(document);
@@ -4543,7 +4518,8 @@ private void writeListLine(XWPFDocument document, DocumentTextStyle style,
measureTheItemAtWordsSize(para, style, lines, measuredColumns.get(numId));
}
}
- writeItemText(para, style, numId != null ? "" : " ".repeat(depth) + marker, label, reading, laidOut);
+ writeItemText(para, style, numId != null ? "" : " ".repeat(depth) + marker, label, numId != null, reading,
+ laidOut);
styleTheMark(para, style);
}
@@ -4656,7 +4632,7 @@ private boolean writeRichListLine(XWPFDocument document, DocumentTextStyle style
java.util.OptionalDouble lineHeight,
java.util.Optional line,
String path, com.demcha.compose.document.node.ListNode measured,
- ItemReading reading) {
+ DocxMarkdown.ItemReading reading) {
warnDroppedInlineRuns(marker.runs(), path);
warnDroppedInlineRuns(item.runs(), path);
line = line.map(listLine -> itemLine(listLine, item.runs()));
@@ -4730,7 +4706,7 @@ private boolean writeRichListLine(XWPFDocument document, DocumentTextStyle style
}
}
} else {
- writeItemText(para, style, "", item.label(), reading, laidOut);
+ writeItemText(para, style, "", item.label(), false, reading, laidOut);
}
makeRoomForPictures(para, pictures);
styleTheMark(para, markStyle);
@@ -7800,8 +7776,8 @@ private String markdownHeadingCut(ParagraphNode node, List pieces, DocumentTextStyle style,
- List lines,
- String whose, String owner) {
+ List lines,
+ String whose, String owner) {
if (pieces == null || style == null || lines.isEmpty()) {
return null;
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
index 48c58206e..0ddfbacc2 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
@@ -34,6 +34,7 @@ class DocxListMarkdownTest {
private static final String MARKS = "its items' markdown marks are written as letters, where the page sets the "
+ "text they mark and drops them";
+ private static final DocumentTextStyle BOLD = DocumentTextStyle.builder().decoration(DocumentTextDecoration.BOLD).build();
@Test
void aListsItemsAreWrittenInThePiecesThePageSetsThem() throws Exception {
@@ -106,26 +107,40 @@ void aListWrittenAsAParagraphPerItemWritesItsMarkerAndIndentBeforeThePieces() th
}
@Test
- void aMarkerThePageSetsInAnotherFaceThanTheListsIsWrittenInIt() throws Exception {
- DocumentTextStyle bold = DocumentTextStyle.builder().decoration(DocumentTextDecoration.BOLD).build();
- // The page's parser sets a bold list's nested marker, read with the item, regular. Written
- // as characters, the marker takes the face the page sets it in.
- Export text = export(true, list -> list.textStyle(bold).markerFor(1, ListMarker.none())
+ void aMarkerWrittenAsCharactersIsWrittenInTheFaceThePageSetsItIn() throws Exception {
+ // The page's parser sets the marker of a bold list's tree of items, read with the item,
+ // regular. Written as characters, the marker takes the face the page sets it in.
+ Export text = export(true, list -> list.textStyle(BOLD).markerFor(1, ListMarker.none())
.addItem("**Lead** item", child -> child.addItem("sub")));
assertThat(text.paragraphWith("Lead").getRuns()).extracting(XWPFRun::text).containsExactly("• ", "Lead", " item");
assertThat(text.paragraphWith("Lead").getRuns()).extracting(XWPFRun::isBold).containsExactly(false, true, false);
assertThat(text.notes()).noneMatch(note -> note.contains("markdown"));
- // Word draws a Word list's marker in the list's face: the item is written as authored.
- Export word = export(true, list -> list.textStyle(bold).addItem("**Lead** item", child -> child.addItem("sub")));
- assertThat(word.paragraphWith("Lead").getNumID()).isNotNull();
- assertThat(word.paragraphWith("Lead").getText()).isEqualTo("**Lead** item");
- assertThat(word.notes()).singleElement().asString().endsWith("; " + MARKS);
- // A flat list sets its marker before the item's text, in the list's face, as Word draws it.
- Export flat = export(true, list -> list.textStyle(bold).items("**Lead** item"));
+ // A flat list sets its marker before the item's text, in the list's face, as Word draws it;
+ // the text the parser keeps every mark of it sets regular, as it sets every piece.
+ Export flat = export(true, list -> list.textStyle(BOLD).items("**Lead** item", "node_js"));
assertThat(flat.paragraphWith("Lead").getRuns()).extracting(XWPFRun::isBold).containsExactly(true, false);
+ assertThat(flat.paragraphWith("node_js").getRuns()).extracting(XWPFRun::isBold).containsExactly(false);
assertThat(flat.notes()).isEmpty();
}
+ @Test
+ void anItemWhoseMarkerWordDrawsInAnotherFaceIsWrittenAsAuthoredAndNamed() throws Exception {
+ // Word draws a Word list's marker in the list's face, where the page sets it regular: the
+ // item is written as authored, and named whatever marks the page keeps of it.
+ Export word = export(true, list -> list.textStyle(BOLD).addItem("**Lead** item",
+ child -> child.addItem("Kot_lin").addItem("Plain")));
+ assertThat(word.paragraphWith("Lead").getNumID()).isNotNull();
+ assertThat(word.paragraphWith("Lead").getText()).isEqualTo("**Lead** item");
+ assertThat(word.paragraphWith("Kot_lin").getRuns()).extracting(XWPFRun::isBold).containsExactly(true);
+ assertThat(word.notes()).singleElement().asString().endsWith("; 2 of its 3 items are written as authored, "
+ + "where the page reads them as markdown and sets their markers in another face than the list's, the "
+ + "face Word draws a list's marker in");
+ Export one = export(true, list -> list.textStyle(BOLD).addItem("snake_case", child -> child.addItem("sub")));
+ assertThat(one.notes()).singleElement().asString().endsWith("; 1 of its 2 items is written as authored, where "
+ + "the page reads it as markdown and sets its marker in another face than the list's, the face Word "
+ + "draws a list's marker in");
+ }
+
@Test
void aLeadThePageReadsAsMarkdownLeavesTheItemWrittenAsAuthoredAndNamed() throws Exception {
// A marker of marks, laid out in a nested item's text, is read with it: the page sets
@@ -169,6 +184,25 @@ void itemsNotMatchedToTheLayoutsAreWrittenAsAuthoredAndNamed() throws Exception
Export export = export(true, list -> list.items("Lead", "**Long** item that runs on and on. ".repeat(30)));
assertThat(export.paragraphWith("Long").getText()).startsWith("**Long**");
assertThat(export.notes()).singleElement().asString().endsWith("; " + MARKS);
+ // A blank item a hangingIndent list draws as a marker alone is a row the export writes none of.
+ Export blank = export(true, list -> list.hangingIndent(true).items("**Java**", "", "Kotlin"));
+ assertThat(blank.paragraphWith("Java").getText()).isEqualTo("**Java**");
+ assertThat(blank.notes()).singleElement().asString().endsWith("; " + MARKS);
+ }
+
+ @Test
+ void aHangingIndentTreeWrittenAsAParagraphPerItemWritesThePiecesAfterItsMarker() throws Exception {
+ // A level with no marker makes it no Word list; its markers stand in a column of their own,
+ // and the pieces are the item's text alone.
+ Export export = export(true, list -> list.hangingIndent(true).markerFor(1, ListMarker.none())
+ .addItem("**Lead** item", child -> child.addItem("*Java* sub")));
+ XWPFParagraph lead = export.paragraphWith("Lead");
+ assertThat(lead.getNumID()).isNull();
+ assertThat(lead.getText()).doesNotContain("*").endsWith("Lead item");
+ assertThat(lead.getRuns()).filteredOn(XWPFRun::isBold).extracting(XWPFRun::text).containsExactly("Lead");
+ assertThat(export.paragraphWith("Java").getRuns()).filteredOn(XWPFRun::isItalic).extracting(XWPFRun::text)
+ .containsExactly("Java");
+ assertThat(export.notes()).noneMatch(note -> note.contains("markdown"));
}
@Test
@@ -206,6 +240,16 @@ void theFacesTheItemsPiecesAreSetInTravelWithTheDocument() throws Exception {
.items("**Java**", "Kotlin"));
assertThat(partXml(flat.document(), "/word/fontTable")).contains(" list.textStyle(boldLato).markerFor(1, ListMarker.none())
+ .addItem("**Lead**", child -> child.addItem("sub")));
+ assertThat(partXml(lead.document(), "/word/fontTable")).contains(" list.textStyle(boldLato).items("* Java", "* Kotlin"));
+ assertThat(partXml(typed.document(), "/word/fontTable")).contains(" texts = exportTexts(flow -> flow
- .addList("**bold** lead stays intact"));
-
- // The session reads markdown, and the item is written as the page sets it: its bold
- // lead's marks dropped, not one of them taken off as a typed marker.
- assertThat(texts).contains("bold lead stays intact");
+ // A session that reads no markdown sets the item as authored, its bold lead whole.
+ try (XWPFDocument document = export(false, flow -> flow.addList("**bold** lead stays intact"))) {
+ assertThat(document.getParagraphs()).extracting(XWPFParagraph::getText)
+ .contains("**bold** lead stays intact");
+ }
+ // One that reads it sets the bold lead's text bold, its marks dropped, and the item is
+ // written so: not one of them was taken off as a typed marker.
+ assertThat(exportTexts(flow -> flow.addList("**bold** lead stays intact")))
+ .contains("bold lead stays intact");
}
@Test
@@ -139,10 +142,16 @@ private static long listItemCount(
private static XWPFDocument export(
Consumer author) throws Exception {
+ return export(true, author);
+ }
+
+ private static XWPFDocument export(boolean markdown,
+ Consumer author) throws Exception {
byte[] docxBytes;
try (DocumentSession session = GraphCompose.document()
.pageSize(595, 842)
.margin(DocumentInsets.of(36))
+ .markdown(markdown)
.create()) {
var flow = session.dsl().pageFlow().name("Flow");
author.accept(flow);
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
index 6ccbc4a5c..dd199a290 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
@@ -1,6 +1,8 @@
package com.demcha.compose.document.backend.semantic.docx;
+import com.demcha.compose.document.dsl.ListBuilder;
import com.demcha.compose.document.layout.payloads.ParagraphLine;
+import com.demcha.compose.document.node.ListMarker;
import com.demcha.compose.document.layout.payloads.ParagraphShapeSpan;
import com.demcha.compose.document.layout.payloads.ParagraphSpan;
import com.demcha.compose.document.layout.payloads.ParagraphTextSpan;
@@ -164,6 +166,30 @@ void piecesAreSplitAtTheEndOfTheLeadTheyOpenWith() {
.as("pieces shorter than the lead").isNull();
}
+ @Test
+ void aListsItemsAreReadAsThePageLaysThemOut() {
+ String indent = Character.toString(0x00A0).repeat(2);
+ // A flat list sets its marker before the item's first line, apart from its text, a typed
+ // marker taken off; with hangingIndent the marker stands in a column of its own.
+ assertThat(DocxMarkdown.items(new ListBuilder().items("* **Java**", "", "Kotlin").build())).containsExactly(
+ new DocxMarkdown.ItemReading("**Java**", "", "• "), new DocxMarkdown.ItemReading("Kotlin", "", "• "));
+ assertThat(DocxMarkdown.items(new ListBuilder().hangingIndent(true).items("**Java**").build()))
+ .containsExactly(new DocxMarkdown.ItemReading("**Java**", "", ""));
+ assertThat(DocxMarkdown.items(new ListBuilder().noMarker().items("**Java**").build()))
+ .containsExactly(new DocxMarkdown.ItemReading("**Java**", "", ""));
+ // A tree of items is laid out in labels, each after its depth's indent and its marker.
+ assertThat(DocxMarkdown.items(new ListBuilder().markerFor(1, ListMarker.none())
+ .addItem("**A**", child -> child.addItem("b_c", grand -> grand.addItem("*d*")))
+ .addItem(rich -> rich.bold("runs")).build())).containsExactly(
+ new DocxMarkdown.ItemReading("• **A**", "• ", ""),
+ new DocxMarkdown.ItemReading(indent + "b_c", indent, ""),
+ new DocxMarkdown.ItemReading(indent + indent + "▪ *d*", indent + indent + "▪ ", ""));
+ // With hangingIndent, its label alone, a typed marker taken off.
+ assertThat(DocxMarkdown.items(new ListBuilder().hangingIndent(true)
+ .addItem("- **A**", child -> child.addItem("b")).build())).containsExactly(
+ new DocxMarkdown.ItemReading("**A**", "", ""), new DocxMarkdown.ItemReading("b", "", ""));
+ }
+
@Test
void whatThePageMayReadAsMarkdownHoldsAMarkOfEmphasisOrCode() {
assertThat(DocxMarkdown.holdsAMark("a *b*")).isTrue();
From 8fc647daa1c13187e856a6c766d677799ae34bef Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Wed, 7 Oct 2026 19:09:27 +0100
Subject: [PATCH 3/3] fix(docx): write a bold Word list's tree item in its
pieces, and count an unmatched list's markers out of its note
A Word list built as a tree of items, with no hangingIndent, has its
items' markers read with them, and the page's parser sets the marker
regular. Such an item was written as authored on the premise that Word
draws the marker in the list's face. Measured in Word 16 on a bold Lato
list, Word draws such a list's bullets regular, whatever the paragraph
mark holds: the bullets' ink matches the page's regular marker. The item
is now written in its pieces, as any other, and the refusal is gone.
A list whose items are not matched to the layout's read its items the
old way: a tree's indent and marker were not read with the item, and the
marker set before each flat item's first line counted towards the marks
its lines hold, so a `*` marker could hide a dropped `**`. It now reads
its items with DocxMarkdown.items and subtracts the markers' marks.
The docs and the CHANGELOG say what is named: an item written as
authored is named where the page drops a mark from it, not where the page
changes only its face or letters no mark is made of.
---
CHANGELOG.md | 24 ++--
.../architecture/backend-capability-matrix.md | 2 +-
docs/recipes/docx-export.md | 26 ++--
render-docx/README.md | 17 +--
.../backend/semantic/docx/DocxMarkdown.java | 20 +--
.../semantic/docx/DocxSemanticBackend.java | 128 +++++++-----------
.../semantic/docx/DocxListMarkdownTest.java | 39 +++---
.../semantic/docx/DocxMarkdownReportTest.java | 2 +-
.../semantic/docx/DocxMarkdownTest.java | 2 +-
9 files changed, 118 insertions(+), 142 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 28966b43c..61ef6e2c0 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -16,28 +16,30 @@ follow semantic versioning; release dates are ISO 8601.
and written in the pieces the page sets it in where those lines hold them, as a paragraph is:
- a flat list's item, after the marker the page sets before its first line;
- an item whose marker stands in a column of its own (`hangingIndent`);
- - an item, at any level, of a list built as a tree of items (`addItem(label, children)`) with
- no `hangingIndent`, which the page lays out after its indent and its marker and reads with
- them: the item's own text is written after them.
- - **The marker stays where it was.** Word draws a Word list's. A list written as a paragraph per
- item writes its marker and nesting indent as characters before the pieces, in the face the
- page sets them in.
+ - an item, at any level, of a list built as a tree of items
+ (`addItem(String, Consumer)`) with no `hangingIndent`, which the page lays out
+ after its indent and its marker and reads with them: the item's own text is written after
+ them.
+ - **The marker stays where it was.** Word draws a Word list's. The page's parser sets the marker
+ it reads with a tree's item regular, a bold list's too, and Word draws such a list's bullets
+ regular (measured in Word 16). A list written as a paragraph per item writes its marker and
+ nesting indent as characters before the pieces, in the face the page sets them in.
- **A heading written taller than its item's line is named**, as a paragraph's is.
- **The font table ships the faces the page sets an item's pieces in**, read off the text the
page reads: a tree's marker with its item, a marker typed before an item taken off.
- - **Still written as authored, and named:**
+ - **Still written as authored**, and named where the page drops a mark from the item:
- the items of a list not matched one by one to the layout's: with no layout, composed in a
table cell, an item run onto the next page, or a `hangingIndent` list with a blank item the
page draws as a marker alone;
- - an item of a Word list built as a tree of items whose marker the page reads with it and sets
- in another face than the list's — the parser sets a bold list's marker regular — since Word
- draws it in the list's; it is named whatever marks the page keeps of it;
- an item of a tree whose marker the parser reads as markdown with it, as `*a*`;
- one the page sets in other letters than its text, as Arabic;
- an item the parser reads into nothing, as `***`: the page sets none of its text, and the
note now says so, where it named the item's marks.
- Across the DOCX fidelity corpus no list item is read as markdown, and the 62 documents are
+ An item written as authored whose marks the page keeps all of is not named where the page
+ changes only its face — `node_js` in a bold list composed in a table cell, set regular — or
+ letters no mark is made of, as an ordered item's number, `1.`; neither is a paragraph's. Across
+ the DOCX fidelity corpus no list item is written otherwise than before, and the 62 documents are
byte-identical. `DocxMarkdown` reads each item as the page lays it out, and `DocxMarkdown.split`
takes a tree's indent and marker off an item's pieces, with unit tests.
diff --git a/docs/architecture/backend-capability-matrix.md b/docs/architecture/backend-capability-matrix.md
index 0941d4f46..3d77a2b17 100644
--- a/docs/architecture/backend-capability-matrix.md
+++ b/docs/architecture/backend-capability-matrix.md
@@ -64,7 +64,7 @@ Payload records live in `core` under
| Capability (payload) | PDF (fixed) | PPTX (fixed) | DOCX (semantic) |
|---|---|---|---|
| Paragraph — pre-wrapped lines, runs, alignment (`ParagraphFragmentPayload`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` (one absolute, wrap-disabled frame per measured line) | ⚠️ semantic paragraphs (`DocxSemanticBackend`) — each run keeps its own style, falling back to the paragraph's when it has none; a centred or right-aligned left-to-right line of its own, of text alone and untracked, that Word sets a point or more wider or narrower at its half-point size has its letters spaced by the difference (`w:spacing`) and its room reckoned from the page's width; a `linkTarget` becomes a `w:hyperlink`, with a relationship for an address or `w:anchor` for one of the document's own anchors, and a run's own link wins over the paragraph's; a paragraph seated off its baseline (`TextVerticalAlign`) has its runs raised or lowered in the line (`w:position`) by the PDF backend's own correction (`ParagraphSeating`), one shift for the paragraph where the page seats each line by its own; Word and LibreOffice stand an exact line's baseline four fifths of the way down it whatever the face, where the page sets it the face's ascent down, so a paragraph whose face puts the two half a point or more apart — Spectral's, not Lato's — has its text moved to the page's baseline in the same position, matched at its middle line (not yet a list item's or a table text cell's; a picture among it moves with it in Word and stays on its own baseline in LibreOffice); lines a container stacks over one another tighter than their face each end halfway between their letters and the next line's (Word draws an exact line's text on screen only inside the line; its PDF export does not cut it), and the last layer of a shape container on one page, where its line runs past the foot, ends at the foot or below its letters; letters two lines share are split halfway so the page does not move, and a stack that holds a picture keeps its lines' own heights; a `bulletOffset` of spaces becomes the paragraph's indent (`w:ind` left, hanging or first line, by `indentStrategy`) in the flow and in cells, not yet over the flow, in an overlay's left-and-right pair, as a badge's initials or in a header or footer; one with letters in it is not written, its wrapped lines still set after the spaces that cover it; an auto-sized paragraph's text is written at its style's size, not the one the page fits it to; a paragraph a session reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, read through the page's own parser, one run a piece in the face, family, colour, tracking and size the page's laid-out lines hold (an auto-sized one's at its style's size), its marks dropped, wherever the lines hold the pieces' letters so; where they are not read, or hold other letters or none (text the parser reads into nothing, which the page sets as nothing), as authored, its marks as letters; a `bookmark(...)` is Word's `HeadingN`, which Word's outline lists by the text of its Word paragraph — an overlay's pair's whole line, one level for both sides — at no level past the ninth. Outside a header or footer, the paragraph's report note (`ParagraphNode`) names each of these where it moves or renames something: the prefix's letters, and the room a path that writes no prefix leaves out where it moves a line; the size written and the size the page fits the text to, to Word's half point; the marks of a paragraph the page read as markdown and the file holds as letters, where its laid-out lines hold fewer of them than its text, not measured where its lines are not read; a markdown heading written taller than the line the page sets it in, which Word cuts on screen; an outline title that is not the text Word lists, a level past the ninth that shares it with another, and the right side's entry where the left holds the line's level |
-| List hanging indent — a marker column and a content column (`ListBuilder.hangingIndent(true)`, `markerGap(...)`) | ✅ marker and content emitted as separate `ParagraphFragmentPayload` fragments at the resolved `markerX` / `contentX` | ✅ the same fragments — the fixed-layout pipeline resolves the geometry before either backend sees it | ⚠️ the top level only. `DocxSemanticBackend` exports a list as a real Word list — `numbering.xml`, `w:numPr` per item, the level carrying the marker — or, with rich items or a drawn marker, as paragraphs; content and nesting are unaffected. With the flag, the top level's marker column is the layout's — the marker's width and `markerGap`, the text and its wrapped lines where the page sets them — where the gap covers what Word may set the marker wider: a picture at its written size, its edges included, or text in the page's face (embedded, or a standard one Word sets in the same widths) grown to its half-point size, half a point clear. A Word list's level then indents and hangs by that column; a list of paragraphs writes the marker, a tab to a stop there, and hangs the item there. Word places content at absolute indents and has no relative-advance primitive, so without the layout's measure the gap could not be honoured; a Word list without the flag that the layout placed and that does not nest takes the page's column too, the spaces the page sets its wrapped lines after, its marker followed by a space (`w:suff`) and an item that wraps measured at Word's half-point size; a list that nests items, a list built as a tree of items (laid out flattened), and a marker the gap does not clear keep the stated column (180 twips, plus 120 per nesting level) — except, in a list of paragraphs, a nested rich item with no marker, which stands where the layout set its text, its measure weighed at Word's half-point sizes, where the layout's items are matched to the list's; a list that nests only such items sets its top level at the page's column too. The report counts, on the list, the items that stand at a stated column, a space past their marker or two spaces a level in, and names a centred or right-aligned list written flush left, a lineSpacing not written where the layout's items are not the list's own and one wraps (in a list composed in a table cell, its wrapping not measured), a continuationIndent not written where an item of a markerless list or a tree of items without the flag wraps or its wrapping is not measured, the rows the page draws as a marker alone for blank items of a flagged list, which are not written, and what items the page reads as markdown lose. An item of plain text the page reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, matched to the lines the page laid it out in and read as the page lays it out (a flat item after the marker the page sets before it; an item of a list built as a tree of items without the flag after the indent and marker the page reads with it), one run a piece in the face and size those lines hold, Word or the file drawing the marker where it did, the font table shipping the faces the page sets the pieces in; it is written as authored, its marks as letters, and named where its list's items are not matched one by one to the layout's (no layout, composed in a table cell, an item run onto the next page, a flagged list with a blank item), where its lines hold other letters, where the parser reads a tree's marker as markdown with the item (`*a*`), where a Word list's tree item has its marker set in another face on the page than the list's (named whatever marks the page keeps), and where the parser reads it into nothing; a markdown heading written taller than its item's line is named |
+| List hanging indent — a marker column and a content column (`ListBuilder.hangingIndent(true)`, `markerGap(...)`) | ✅ marker and content emitted as separate `ParagraphFragmentPayload` fragments at the resolved `markerX` / `contentX` | ✅ the same fragments — the fixed-layout pipeline resolves the geometry before either backend sees it | ⚠️ the top level only. `DocxSemanticBackend` exports a list as a real Word list — `numbering.xml`, `w:numPr` per item, the level carrying the marker — or, with rich items or a drawn marker, as paragraphs; content and nesting are unaffected. With the flag, the top level's marker column is the layout's — the marker's width and `markerGap`, the text and its wrapped lines where the page sets them — where the gap covers what Word may set the marker wider: a picture at its written size, its edges included, or text in the page's face (embedded, or a standard one Word sets in the same widths) grown to its half-point size, half a point clear. A Word list's level then indents and hangs by that column; a list of paragraphs writes the marker, a tab to a stop there, and hangs the item there. Word places content at absolute indents and has no relative-advance primitive, so without the layout's measure the gap could not be honoured; a Word list without the flag that the layout placed and that does not nest takes the page's column too, the spaces the page sets its wrapped lines after, its marker followed by a space (`w:suff`) and an item that wraps measured at Word's half-point size; a list that nests items, a list built as a tree of items (laid out flattened), and a marker the gap does not clear keep the stated column (180 twips, plus 120 per nesting level) — except, in a list of paragraphs, a nested rich item with no marker, which stands where the layout set its text, its measure weighed at Word's half-point sizes, where the layout's items are matched to the list's; a list that nests only such items sets its top level at the page's column too. The report counts, on the list, the items that stand at a stated column, a space past their marker or two spaces a level in, and names a centred or right-aligned list written flush left, a lineSpacing not written where the layout's items are not the list's own and one wraps (in a list composed in a table cell, its wrapping not measured), a continuationIndent not written where an item of a markerless list or a tree of items without the flag wraps or its wrapping is not measured, the rows the page draws as a marker alone for blank items of a flagged list, which are not written, and what items the page reads as markdown lose. An item of plain text the page reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, matched to the lines the page laid it out in and read as the page lays it out (a flat item after the marker the page sets before it; a flagged item as its text alone; an item of a list built as a tree of items without the flag after the indent and marker the page reads with it), one run a piece in the face, family, colour, tracking and size those lines hold, Word or the file drawing the marker where it did (Word draws a tree's bullets regular, as the parser sets the marker it reads), the font table shipping the faces the page sets the pieces in; it is written as authored, its marks as letters, where its list's items are not matched one by one to the layout's (no layout, composed in a table cell, an item run onto the next page, a flagged list with a blank item), where its lines hold other letters, where the parser reads a tree's marker as markdown with the item (`*a*`), and where the parser reads it into nothing, and named where the page drops a mark from it or sets none of its text — not where it changes only its face or letters no mark is made of (`1.`); a markdown heading written taller than its item's line is named |
| Inline code/badge chips (`InlineBackground` on text spans) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ⚠️ `DocxSemanticBackend` — the fill becomes the run's own `w:shd`, in a paragraph and in a list item alike, so a badge still reads as a badge. What Word has no way to say is the shape: shading covers the glyph box, so the corner radius and the padding above and below the letters are not in the file, and the export records them. The padding beside the letters is written as the room it takes (`spaceAfterTheLastLetter`): character spacing after the chip's last letter, shaded with it, and after the letter before the chip, unshaded. A chip opening its line or following a picture has no letter before it, so its left padding is not in the file; no space is written after right-to-left letters or after a symbol or emoji. The export records, chip by chip, how each side was written. LibreOffice sets no spacing after a line's last letter, so it does not apply the right padding of a chip that ends a line. A `w:shd` fill is opaque, so a translucent chip is flattened first against what Word paints underneath it — the paragraph's shading, the cell's, or else the colour the page paints under the paragraph, a page background included — so the chip agrees with the file it is in and shows the colour the PDF shows. It stops being translucent, and that is recorded with the rest |
| Inline images (`ParagraphImageSpan`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ✅ `DocxSemanticBackend.writeInlinePicture` (a picture in its own run where it sits among the words, at its size, inside the run's or the paragraph's link; raised or lowered by `w:position` to where the page's alignment and `baselineOffset` put it, from the layout's measure of the paragraph's first line — in a list, the list's text on a line as tall as the item's own tallest picture; LibreOffice ignores `w:position` on a picture and stands it on the baseline, so a picture the export draws itself (icon, emoji, shape) that the page raises carries the rise as transparent rows and needs no `w:position`, while one the page lowers stands in LibreOffice higher than on the page by as much as the page lowers it — up to the text's descent for a centred icon as tall as its line; the editor clips a picture to an exact line height, so a paragraph holding a picture that leaves its text — past the ascent or the descent, in Word's placement or on the baseline — has its lines written at least the height the picture reaches, grown by the editor rather than clipped, every line of the paragraph since Word has one line height for it, and each as tall as the editor's font makes it — for 14pt text about 2.5pt taller than the page's in LibreOffice; a picture inside its text in both editors keeps the exact height; a paragraph of one line of text in a Word paragraph of its own, with room above for its pictures' reach, keeps an exact line at the page's height of it, the pictures set in it where the page puts them in Word and what their ink reaches past it taken from the gaps around it, and in LibreOffice a lowered picture there stands higher and loses what passes the line's top; its description is the text it stands for or empty) |
| Inline vector shapes (`ParagraphShapeSpan`) | ✅ `PdfParagraphFragmentRenderHandler` | ⚠️ `PptxParagraphFragmentRenderHandler` + `PptxInlineGeometry` (distinct per-corner radii render with the top-left radius — single-adjust preset) | ⚠️ `DocxSemanticBackend.writeInlinePicture` + `DocxShapePictures` (a transparent PNG drawn by the shared `InlineSvgRasters` from the outline, fill and stroke — every outline kind, each layer centred in the run's box — placed as an inline picture is; the picture takes as far as the stroked ink reaches past the outline — half the stroke on an edge, more at a sharp corner's miter — and a pixel on each side, measured side by side, and is lowered by what it takes below, so no edge is cut and a shape takes that much more room in the line; a list marker that draws a disc is its picture; at the top level of a `hangingIndent(true)` list that does not nest it is followed by a tab to where the layout starts the item's text, the item's lines hanging there, when the picture clears that stop, else by a space) |
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 1eda66158..a551e78d0 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -468,18 +468,20 @@ list's item is read after the marker the page sets before its first line; an ite
stands in a column of its own, as its text alone; an item, at any level, of a list built as a
tree of items without `hangingIndent`, which the page lays out after its indent and its marker
and reads with them, as its own text after them. The marker stays where it was: Word draws a
-Word list's, and a list of paragraphs writes its marker and nesting indent as characters before
-the pieces, in the face the page sets them in. The font table ships the faces the page sets the
-pieces in. The report names a heading written taller than its item's line, which Word cuts on
-screen. An item is written as authored, its marks as letters, and named where its list's items
-are not matched one by one to the layout's — with no layout, composed in a table cell, an item
-run onto the next page, a `hangingIndent` list with a blank item the page draws as a marker
-alone — where its lines hold other letters than its pieces, as Arabic, or where the parser reads
-the marker of a tree's item as markdown with it, as `*a*`. An item of a Word list built as a tree
-whose marker the page sets in another face than the list's — its parser sets a bold list's marker
-regular — is written as authored too, since Word draws the marker in the list's, and named
-whatever marks the page keeps of it. An item the parser reads into nothing, as `***`, is written
-as authored, and named as an item the page sets none of the text of.
+Word list's — the parser sets the marker it reads with a tree's item regular, a bold list's too,
+and Word draws such a list's bullets regular (measured in Word 16) — and a list of paragraphs
+writes its marker and nesting indent as characters before the pieces, in the face the page sets
+them in. The font table ships the faces the page sets the pieces in. The report names a heading
+written taller than its item's line, which Word cuts on screen. An item is written as authored,
+its marks as letters, where its list's items are not matched one by one to the layout's — with
+no layout, composed in a table cell, an item run onto the next page, a `hangingIndent` list with
+a blank item the page draws as a marker alone — where its lines hold other letters than its
+pieces, as Arabic, or where the parser reads the marker of a tree's item as markdown with it, as
+`*a*`; the report names it where the page drops a mark from it. An item the parser reads into
+nothing, as `***`, is written as authored, and named as an item the page sets none of the text
+of. An item written as authored whose marks the page keeps all of is not named where the page
+changes only its face — `node_js` in a bold list in a table cell, set regular — or letters no
+mark is made of, as an ordered item's number, `1.`; nor is a paragraph's.
## What a panel keeps and loses
diff --git a/render-docx/README.md b/render-docx/README.md
index f608473b2..a92d3694f 100644
--- a/render-docx/README.md
+++ b/render-docx/README.md
@@ -127,14 +127,15 @@ What is not written — each one is named in the export report
- the marker column and `markerGap` of an item at a stated column — a list that nests, or a gap
too narrow for its marker — while a flat hanging-indent list keeps the page's column;
- a row the page draws as a marker alone, for a blank item;
- - the marks of items the page reads as markdown, where they are written as letters: an item is
- written as the page sets it — its marks dropped, each piece in the page's face and size —
- wherever its lines show how, and as authored where its list's items are not matched one by one
- to the layout's (no layout, a cell, an item run onto the next page, a hanging-indent list with a
- blank item), its lines hold other letters, the parser reads a tree's marker as markdown with it
- (`*a*`), or the page sets none of its text; a Word list's item whose marker the page sets in
- another face than the list's, written as authored whatever marks the page keeps; and a
- markdown heading written taller than its line.
+ - the marks of items the page reads as markdown, where they are written as letters and the page
+ drops one: an item is written as the page sets it — its marks dropped, each piece in the
+ page's face and size — wherever its lines show how, and as authored where its list's items are
+ not matched one by one to the layout's (no layout, a cell, an item run onto the next page, a
+ hanging-indent list with a blank item), its lines hold other letters, the parser reads a
+ tree's marker as markdown with it (`*a*`), or the page sets none of its text, which is named
+ as such; and a markdown heading written taller than its line. An item written as authored
+ whose marks the page keeps is not named where the page changes only its face or letters no
+ mark is made of (`1.`).
- **What a paragraph's own fields set where Word cannot hold it**, named in the export report on
the paragraph outside a header or footer (a page zone's are named on the zone):
- the size an auto-sized paragraph's text is fitted to, where Word, to its half point, holds it
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
index 7d7642f4d..2e84639ae 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
@@ -46,6 +46,13 @@ final class DocxMarkdown {
/** How far apart two sizes may be and still be the same, in points. */
private static final double SIZE_CLEARANCE = 0.01;
+ /**
+ * The indent a list with no {@code hangingIndent} sets an item of a tree of items in, a level
+ * at a time: two no-break spaces ({@code TextFlowSupport}, where it flattens the tree into
+ * labels), which the parser reads as letters.
+ */
+ private static final String NESTED_ITEM_INDENT = Character.toString(0x00A0).repeat(2);
+
private DocxMarkdown() {
}
@@ -66,13 +73,6 @@ static boolean mayRead(ParagraphNode node) {
return (node.inlineRuns() == null || node.inlineRuns().isEmpty()) && holdsAMark(node.text());
}
- /**
- * The indent a list with no {@code hangingIndent} sets an item of a tree of items in, a level
- * at a time: two no-break spaces ({@code TextFlowSupport}, where it flattens the tree into
- * labels), which the parser reads as letters.
- */
- private static final String NESTED_ITEM_INDENT = Character.toString(0x00A0).repeat(2);
-
/**
* A list item's text as the page lays it out, and reads it where its session reads markdown.
*
@@ -233,10 +233,10 @@ private static void add(List pieces, String text, DocumentTextStyle style
* Pieces read off text that opens with a lead the file writes apart from them, split at the
* lead's end.
*
- * @param lead the style the page sets the lead in, {@code null} where there is no lead
- * @param after the pieces after the lead
+ * @param leadStyle the style the page sets the lead in, {@code null} where there is no lead
+ * @param after the pieces after the lead
*/
- record Split(DocumentTextStyle lead, List after) {
+ record Split(DocumentTextStyle leadStyle, List after) {
}
/**
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 3fbd97a97..db52722bd 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -3917,12 +3917,13 @@ private static String levelText(com.demcha.compose.document.node.ListMarker mark
}
/**
- * Semantic list mapping: each item becomes a marker-prefixed paragraph in
- * the list's text style. Flat items run through the same
- * {@code ListMarker.normalizeItemText} step as fixed-layout rendering
- * (author-typed markers stripped, blank items skipped); nested items
- * indent two spaces per depth and use their own marker when one is set,
- * falling back to {@code ListMarker.defaultForDepth} otherwise.
+ * Semantic list mapping: each item becomes a paragraph, a Word list's or one
+ * with its marker as characters, in the list's text style, or in the pieces the
+ * page sets it in where it reads the item as markdown ({@link #writeItemText}).
+ * Flat items run through the same {@code ListMarker.normalizeItemText} step as
+ * fixed-layout rendering (author-typed markers stripped, blank items skipped);
+ * nested items indent two spaces per depth and use their own marker when one is
+ * set, falling back to {@code ListMarker.defaultForDepth} otherwise.
*/
private void writeList(XWPFDocument document,
com.demcha.compose.document.node.ListNode list) {
@@ -4033,12 +4034,10 @@ private List listLost(com.demcha.compose.document.node.ListNode list,
* What a list's items lose where the page reads them as markdown, each item matched to the
* lines the page laid it out in. An item of plain text holding a mark of emphasis or code is
* written in the pieces the page sets it in where its lines say so ({@link #itemPieces}); a
- * heading among them in a line Word cuts it in is named ({@link #headingCut}). One whose marker
- * Word draws in the list's face, where the page sets it in another, is written as authored and
- * named, whatever marks the page keeps of it. One the page reads into nothing is written as
- * authored and named; any other written as authored, its marks and all, is named where its
- * lines, less the prefix the page sets before its first line, hold fewer marks than it
- * ({@link #marksDropped}).
+ * heading among them in a line Word cuts it in is named ({@link #headingCut}). One the page
+ * reads into nothing is written as authored and named; any other written as authored, its
+ * marks and all, is named where its lines, less the prefix the page sets before its first line,
+ * hold fewer marks than it ({@link #marksDropped}).
*
* @param items how many items the list writes
* @param written its items of plain text as written
@@ -4054,17 +4053,8 @@ private List itemsMarkdownLost(com.demcha.compose.document.node.ListNode
break;
}
}
- long refused = written.stream().filter(ItemWritten::refused).count();
- if (refused > 0) {
- boolean one = refused == 1;
- lost.add(refused + " of its " + items + " items " + (one ? "is" : "are")
- + " written as authored, where the page reads " + (one ? "it" : "them")
- + " as markdown and sets " + (one ? "its marker" : "their markers")
- + " in another face than the list's, the face Word draws a list's marker in");
- }
List asAuthored = written.stream()
- .filter(item -> item.pieces() == null && !item.refused() && DocxMarkdown.holdsAMark(item.reading().text()))
- .toList();
+ .filter(item -> item.pieces() == null && DocxMarkdown.holdsAMark(item.reading().text())).toList();
List nothing = asAuthored.stream().filter(item -> setsNoneOfIt(item, list.textStyle())).toList();
if (!nothing.isEmpty()) {
boolean one = nothing.size() == 1;
@@ -4119,32 +4109,25 @@ private static long lettersIn(String text) {
* item of a {@code hangingIndent} list the page draws as a marker alone — and so each written
* as authored, as a paragraph's text is ({@link #markdownLost}): an item of plain text holding a
* mark of emphasis or code is set as markdown, its marks dropped. Its text is read as the page
- * lays it out, a marker typed before it taken off
- * ({@link com.demcha.compose.document.node.ListMarker#normalizeItemText}); its laid-out lines
- * are the items', without a marker laid out on its own. A marker laid out in an item's line,
- * and a rich item's runs, hold every mark they had.
+ * lays it out ({@link DocxMarkdown#items}): a marker typed before it taken off, a tree's indent
+ * and marker read with it. Its laid-out lines are the items', without a marker laid out on its
+ * own, and less the marks of the marker the page sets before a flat item's first line. A rich
+ * item's runs hold every mark they had.
*/
private List itemsMarkdownUnmatched(com.demcha.compose.document.node.ListNode list) {
- List plain = new ArrayList<>();
- List all = new ArrayList<>();
- if (list.nestedItems().isEmpty()) {
- for (String item : list.items()) {
- String text = com.demcha.compose.document.node.ListMarker.normalizeItemText(item, list.normalizeMarkers());
- plain.add(text);
- all.add(text);
- }
- } else {
- collectItemTexts(list.nestedItems(), list.normalizeMarkers(), plain, all);
- }
+ List readings = DocxMarkdown.items(list);
+ List plain = readings.stream().map(DocxMarkdown.ItemReading::text).toList();
if (plain.stream().noneMatch(DocxMarkdown::holdsAMark)) {
return List.of();
}
+ List all = new ArrayList<>(plain);
+ collectRichTexts(list.nestedItems(), all);
List lines = new ArrayList<>();
for (DocxLayoutMetrics.ItemText item : layout.itemLines(list)) {
lines.addAll(item.lines());
}
String marks = marksDropped("its items'", lines, all.stream().mapToInt(DocxSemanticBackend::markdownMarksIn).sum(),
- 0, plain);
+ readings.stream().mapToInt(reading -> markdownMarksIn(reading.prefix())).sum(), plain);
return marks == null ? List.of() : List.of(marks);
}
@@ -4154,12 +4137,10 @@ private List itemsMarkdownUnmatched(com.demcha.compose.document.node.Lis
* @param reading its text as the page reads it
* @param lines the lines the page laid it out in, empty where its list's items are not matched
* @param pieces the pieces it was written in, {@code null} where it was written as authored
- * @param refused whether it was written as authored though its lines hold the pieces the page
- * sets it in: Word draws its marker, which the page sets in another face
*/
private record ItemWritten(DocxMarkdown.ItemReading reading,
List lines,
- List pieces, boolean refused) {
+ List pieces) {
}
/**
@@ -4173,7 +4154,7 @@ private record ItemWritten(DocxMarkdown.ItemReading reading,
*/
private static DocxMarkdown.Split itemPieces(DocxMarkdown.ItemReading reading, DocumentTextStyle style,
DocxLayoutMetrics.ItemText laidOut) {
- if (reading == null || laidOut == null) {
+ if (laidOut == null) {
return null;
}
List pieces = pagePieces(reading.text(), style, laidOut.lines(), reading.prefix(), false);
@@ -4184,25 +4165,20 @@ private static DocxMarkdown.Split itemPieces(DocxMarkdown.ItemReading reading, D
* Writes an item's text: as authored in one run, or in the pieces the page sets it in, and
* records it for the list's note.
*
- * @param before what the file writes before the item's own text — its indent and
- * marker, where Word does not draw the marker — in the lead's style
- * where the page sets the lead in the pieces
- * @param label the item's text as authored
- * @param wordDrawsMarker whether Word draws the item's marker, a Word list's
- * @param laidOut how the layout set the item, {@code null} where its list's items are
- * not matched
+ * Where Word draws the marker of a tree's item the page reads with it, it draws it as the
+ * page's parser sets it: a list built as a tree of items without {@code hangingIndent} states
+ * no face on its levels, and Word draws their bullets regular, a bold list's included
+ * (measured in Word 16), where the parser sets the marker it reads regular.
+ *
+ * @param before what the file writes before the item's own text — its indent and marker,
+ * where Word does not draw the marker — in the lead's style where the page sets
+ * the lead in the pieces
+ * @param label the item's text as authored
+ * @param laidOut how the layout set the item, {@code null} where its list's items are not matched
*/
private void writeItemText(XWPFParagraph para, DocumentTextStyle style, String before, String label,
- boolean wordDrawsMarker, DocxMarkdown.ItemReading reading,
- DocxLayoutMetrics.ItemText laidOut) {
+ DocxMarkdown.ItemReading reading, DocxLayoutMetrics.ItemText laidOut) {
DocxMarkdown.Split split = itemPieces(reading, style, laidOut);
- // Word draws a Word list's marker in the list's style. Where the page sets the marker it
- // reads with the item in another — its parser sets a bold list's marker regular — the item
- // is written as authored, and named.
- boolean refused = split != null && wordDrawsMarker && split.lead() != null && !split.lead().equals(style);
- if (refused) {
- split = null;
- }
if (split == null) {
XWPFRun run = para.createRun();
applyStyle(run, style);
@@ -4210,7 +4186,7 @@ private void writeItemText(XWPFParagraph para, DocumentTextStyle style, String b
} else {
if (!before.isEmpty()) {
XWPFRun lead = para.createRun();
- applyStyle(lead, split.lead() != null ? split.lead() : style);
+ applyStyle(lead, split.leadStyle() != null ? split.leadStyle() : style);
setTextBrokenAtLines(lead, before);
}
for (DocxMarkdown.Piece piece : split.after()) {
@@ -4219,27 +4195,17 @@ private void writeItemText(XWPFParagraph para, DocumentTextStyle style, String b
setTextBrokenAtLines(run, piece.text());
}
}
- if (reading != null) {
- listItemsWritten.add(new ItemWritten(reading, laidOut == null ? List.of() : laidOut.lines(),
- split == null ? null : split.after(), refused));
- }
+ listItemsWritten.add(new ItemWritten(reading, laidOut == null ? List.of() : laidOut.lines(),
+ split == null ? null : split.after()));
}
- /**
- * Every item's text in a tree of items, as the page lays it out: a plain item's label, a marker
- * typed before it taken off, and a rich item's runs' text too.
- */
- private static void collectItemTexts(List items, boolean normalizeMarkers,
- List plain, List all) {
+ /** The text of every item of runs in a tree of items, which the page lays out as it stands. */
+ private static void collectRichTexts(List items, List into) {
for (com.demcha.compose.document.node.ListItem item : items) {
- if (item.runs().isEmpty()) {
- String text = com.demcha.compose.document.node.ListMarker.normalizeItemText(item.label(), normalizeMarkers);
- plain.add(text);
- all.add(text);
- } else {
- all.add(InlineRun.plainText(item.runs()));
+ if (item.isRich()) {
+ into.add(InlineRun.plainText(item.runs()));
}
- collectItemTexts(item.children(), normalizeMarkers, plain, all);
+ collectRichTexts(item.children(), into);
}
}
@@ -4490,9 +4456,8 @@ private int writeNestedItem(XWPFDocument document,
* text the export used before Word numbering existed here.
*
* Its text is written in the pieces the page sets it in where the page reads it as
- * markdown ({@link #writeItemText}). The paragraph's mark keeps the list's style whatever
- * piece ends the item: Word draws a list's marker in the mark's style, and the page draws it
- * in the list's.
+ * markdown ({@link #writeItemText}); the paragraph's mark keeps the list's style whatever
+ * piece ends the item.
*
* @param marker the marker written as characters before the item's text, after the nesting
* indent, where Word does not draw it
@@ -4518,8 +4483,7 @@ private void writeListLine(XWPFDocument document, DocumentTextStyle style,
measureTheItemAtWordsSize(para, style, lines, measuredColumns.get(numId));
}
}
- writeItemText(para, style, numId != null ? "" : " ".repeat(depth) + marker, label, numId != null, reading,
- laidOut);
+ writeItemText(para, style, numId != null ? "" : " ".repeat(depth) + marker, label, reading, laidOut);
styleTheMark(para, style);
}
@@ -4706,7 +4670,7 @@ private boolean writeRichListLine(XWPFDocument document, DocumentTextStyle style
}
}
} else {
- writeItemText(para, style, "", item.label(), false, reading, laidOut);
+ writeItemText(para, style, "", item.label(), reading, laidOut);
}
makeRoomForPictures(para, pictures);
styleTheMark(para, markStyle);
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
index 0ddfbacc2..31f93ac43 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxListMarkdownTest.java
@@ -124,21 +124,20 @@ void aMarkerWrittenAsCharactersIsWrittenInTheFaceThePageSetsItIn() throws Except
}
@Test
- void anItemWhoseMarkerWordDrawsInAnotherFaceIsWrittenAsAuthoredAndNamed() throws Exception {
- // Word draws a Word list's marker in the list's face, where the page sets it regular: the
- // item is written as authored, and named whatever marks the page keeps of it.
+ void aBoldWordListsTreeItemsAreWrittenInThePiecesThePageSetsThemIn() throws Exception {
+ // The page's parser sets the marker of a bold list's tree of items, read with the item,
+ // regular, and Word draws the bullets of such a list regular: the item's own text is
+ // written in its pieces, a mark the parser keeps set regular too.
Export word = export(true, list -> list.textStyle(BOLD).addItem("**Lead** item",
child -> child.addItem("Kot_lin").addItem("Plain")));
- assertThat(word.paragraphWith("Lead").getNumID()).isNotNull();
- assertThat(word.paragraphWith("Lead").getText()).isEqualTo("**Lead** item");
- assertThat(word.paragraphWith("Kot_lin").getRuns()).extracting(XWPFRun::isBold).containsExactly(true);
- assertThat(word.notes()).singleElement().asString().endsWith("; 2 of its 3 items are written as authored, "
- + "where the page reads them as markdown and sets their markers in another face than the list's, the "
- + "face Word draws a list's marker in");
- Export one = export(true, list -> list.textStyle(BOLD).addItem("snake_case", child -> child.addItem("sub")));
- assertThat(one.notes()).singleElement().asString().endsWith("; 1 of its 2 items is written as authored, where "
- + "the page reads it as markdown and sets its marker in another face than the list's, the face Word "
- + "draws a list's marker in");
+ XWPFParagraph lead = word.paragraphWith("Lead");
+ assertThat(lead.getNumID()).isNotNull();
+ assertThat(lead.getRuns()).extracting(XWPFRun::text).containsExactly("Lead", " item");
+ assertThat(lead.getRuns()).extracting(XWPFRun::isBold).containsExactly(true, false);
+ assertThat(word.paragraphWith("Kot_lin").getRuns()).extracting(XWPFRun::isBold).containsExactly(false);
+ // An item the page does not read keeps the list's face.
+ assertThat(word.paragraphWith("Plain").getRuns()).extracting(XWPFRun::isBold).containsExactly(true);
+ assertThat(word.notes()).noneMatch(note -> note.contains("markdown"));
}
@Test
@@ -153,7 +152,7 @@ void aLeadThePageReadsAsMarkdownLeavesTheItemWrittenAsAuthoredAndNamed() throws
@Test
void aHeadingInAnItemIsWrittenAndNamedWhereWordCutsIt() throws Exception {
- Export export = export(true, list -> list.items("# Title *x*", "Kotlin"));
+ Export export = export(true, list -> list.items("# Title *x*", "Kotlin", "# Other *y*"));
XWPFRun title = export.paragraphWith("Title").getRuns().get(0);
assertThat(title.text()).isEqualTo("Title *x*");
assertThat(title.isBold()).isTrue();
@@ -188,6 +187,14 @@ void itemsNotMatchedToTheLayoutsAreWrittenAsAuthoredAndNamed() throws Exception
Export blank = export(true, list -> list.hangingIndent(true).items("**Java**", "", "Kotlin"));
assertThat(blank.paragraphWith("Java").getText()).isEqualTo("**Java**");
assertThat(blank.notes()).singleElement().asString().endsWith("; " + MARKS);
+ // A marker of marks the page sets before each item's first line is not counted for the items.
+ Export marked = export(true, list -> list.marker("*").items("A", "B", "C",
+ "**Long** " + "item that runs on and on. ".repeat(40)));
+ assertThat(marked.notes()).singleElement().asString().endsWith("; " + MARKS);
+ // An item of runs keeps every mark it holds, on the page and in the count alike.
+ Export rich = export(true, list -> list.hangingIndent(true).addItem(runs -> runs.plain("x ** y ** z"))
+ .addItem("**Long** " + "item that runs on and on. ".repeat(40), child -> { }));
+ assertThat(rich.notes()).singleElement().asString().endsWith("; " + MARKS);
}
@Test
@@ -222,8 +229,8 @@ void anItemThePageSetsInOtherLettersIsWrittenAsAuthoredAndNamed() throws Excepti
.items("مرحبا **بالعالم**"));
assertThat(export.document().getDocument().xmlText()).contains("**");
assertThat(export.notes()).singleElement().asString().endsWith("; " + MARKS);
- // The marker the page sets before the item's first line holds marks of its own, which are
- // not the item's: its lines hold as many as the item, less the marker's none.
+ // The marker the page sets before the item's first line holds marks of its own, which the
+ // page keeps: its lines hold as many marks as the item, and none of them the item's.
Export marked = export(true, list -> list.marker("**")
.textStyle(DocumentTextStyle.builder().fontName(FontName.AMIRI).build()).items("مرحبا *بالعالم*"));
assertThat(marked.notes()).singleElement().asString().endsWith("; " + MARKS);
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
index 0fcdef74d..d473fb764 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
@@ -76,7 +76,7 @@ void aParagraphThePageSetsAsAuthoredIsNotNamed() throws Exception {
}
@Test
- void aListsItemsThePageReadsAsMarkdownAreWrittenSoAndNotNamed() throws Exception {
+ void aListsItemsThePageReadsAsMarkdownAreNotNamed() throws Exception {
assertThat(listNotes(true, page -> page.addList(list -> list.name("Skills").items("**Java** lead", "Kotlin"))))
.isEmpty();
assertThat(listNotes(true, page -> page.addList(list -> list.name("Skills").hangingIndent(true)
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
index dd199a290..275d68395 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
@@ -142,7 +142,7 @@ void piecesAreSplitAtTheEndOfTheLeadTheyOpenWith() {
String indent = Character.toString(0x00A0).repeat(2);
List pieces = DocxMarkdown.read(indent + "◦ **Java** lead", BODY);
DocxMarkdown.Split split = DocxMarkdown.split(pieces, indent + "◦ ");
- assertThat(split.lead()).isEqualTo(BODY);
+ assertThat(split.leadStyle()).isEqualTo(BODY);
assertThat(split.after()).containsExactly(piece("Java", DocumentTextDecoration.BOLD, 10),
piece(" lead", DocumentTextDecoration.DEFAULT, 10));
// The lead may end inside a piece and span pieces of one style.