diff --git a/CHANGELOG.md b/CHANGELOG.md index 06ab01df..014b0467 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) ### Added +- New parser for DOCX format + ### Changed ### Fixed diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index 29705898..4c3634f3 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -27,6 +27,9 @@ koin-ktor = "4.2.2" ktor = "3.5.2" stately = "2.1.0" +# examples +poi-ooxml = "5.5.1" + # wfd-xml spock = "2.4-groovy-5.0" xmlunit = "2.13.0" @@ -104,6 +107,8 @@ groovy-json = { module = "org.apache.groovy:groovy-json", version.ref = "groovy" junit-bom = { module = "org.junit:junit-bom", version.ref = "junit-jupiter" } junit-jupiter-api = { module = "org.junit.jupiter:junit-jupiter-api", version.ref = "junit-jupiter" } junit-jupiter-engine = { module = "org.junit.jupiter:junit-jupiter-engine", version.ref = "junit-jupiter" } +poi-ooxml = { module = "org.apache.poi:poi-ooxml", version.ref = "poi-ooxml" } +poi-ooxml-full = { module = "org.apache.poi:poi-ooxml-full", version.ref = "poi-ooxml" } # wfd-xml spock-core = { module = "org.spockframework:spock-core", version.ref = "spock" } diff --git a/migration-examples/build.gradle.kts b/migration-examples/build.gradle.kts index fcac8f56..994e113f 100644 --- a/migration-examples/build.gradle.kts +++ b/migration-examples/build.gradle.kts @@ -45,6 +45,8 @@ dependencies { implementation(libs.groovy.json) implementation(libs.jackson.databind) implementation(libs.slf4j.api) + implementation(libs.poi.ooxml) + implementation(libs.poi.ooxml.full) testImplementation(platform(libs.junit.bom)) testImplementation(libs.junit.jupiter) diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy new file mode 100644 index 00000000..bb81c901 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy @@ -0,0 +1,15 @@ +//! --- +//! displayName: Parse DOCX +//! category: Parser +//! description: Parses input DOCX files specified in the project settings, translates their contents into the migration model and stores the resulting objects in the database. Persisted mappings are not applied. +//! sourceFormat: DOCX +//! --- +package com.quadient.migration.example.docx + +import static com.quadient.migration.example.common.util.InitMigration.initMigration +import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles + +def migration = initMigration(this.binding) + +println("\nStarting Parse step...\n") +parseDocxFiles(migration) diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy new file mode 100644 index 00000000..5bcfe12d --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy @@ -0,0 +1,17 @@ +//! --- +//! displayName: Parse DOCX and Apply Mapping +//! category: Parser +//! description: Parses input DOCX files specified in the project settings, translates their contents into the migration model, stores the resulting objects in the database and applies persisted mappings. +//! sourceFormat: DOCX +//! --- +package com.quadient.migration.example.docx + +import static com.quadient.migration.example.common.util.InitMigration.initMigration +import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles + +def migration = initMigration(this.binding) + +println("\nStarting Parse step...\n") +parseDocxFiles(migration) +println("\nApplying persisted mapping...\n") +migration.mappingRepository.applyAll() diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy new file mode 100644 index 00000000..214f1600 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy @@ -0,0 +1,232 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef +import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder +import com.quadient.migration.shared.ImageOptions +import com.quadient.migration.shared.Position +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFPictureData +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.xmlbeans.XmlCursor +import org.apache.xmlbeans.XmlObject +import org.openxmlformats.schemas.drawingml.x2006.picture.CTPicture +import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTP +import org.w3c.dom.Node +import org.xml.sax.InputSource + +import javax.xml.parsers.DocumentBuilder +import javax.xml.parsers.DocumentBuilderFactory +import javax.xml.transform.OutputKeys +import javax.xml.transform.Transformer +import javax.xml.transform.TransformerFactory +import javax.xml.transform.dom.DOMSource +import javax.xml.transform.stream.StreamResult + +import static com.quadient.migration.example.docx.parser.DocxBlocks.blockName +import static com.quadient.migration.example.docx.parser.DocxBlocks.upsertBlock +import static com.quadient.migration.example.docx.parser.DocxImages.registerImageData +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseParagraph +import static com.quadient.migration.example.docx.util.DocxUtils.emuToSize +import static com.quadient.migration.example.docx.util.DocxUtils.extractParagraphText + +class DocxAnchoredAreas { + private static final String WP_NS = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing" + private static final String W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + private static final String WPS_SHAPE_URI = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape" + private static final String PIC_NS = "http://schemas.openxmlformats.org/drawingml/2006/picture" + private static final double BACKGROUND_COVERAGE_THRESHOLD = 0.6 + + final Map> backgroundAreasByPage = [:] + final Map> floatingAreasByPage = [:] + final Set consumedEmbedIds = [] + private final Set backgroundEmbedIds = [] + private int textBoxes = 0 + + private final Migration migration + private final XWPFDocument doc + private final String fileName + private final DocumentBuilder documentBuilder + private final TransformerFactory transformerFactory = TransformerFactory.newInstance() + + private DocxAnchoredAreas(Migration migration, XWPFDocument doc, String fileName) { + this.migration = migration + this.doc = doc + this.fileName = fileName + DocumentBuilderFactory dbf = DocumentBuilderFactory.newInstance() + dbf.namespaceAware = true + this.documentBuilder = dbf.newDocumentBuilder() + } + + static DocxAnchoredAreas extract(Migration migration, XWPFDocument doc, String fileName, List pages) { + DocxAnchoredAreas result = new DocxAnchoredAreas(migration, doc, fileName) + // Backgrounds are resolved over the whole document first so the same picture is never emitted as floating too. + pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addBackgroundArea(it, page) } } + pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) } } + return result + } + + private void eachUniqueAnchor(DocxPage page, Closure handler) { + Set seenDrawingIds = [] + page.paragraphs().each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + findAnchors(run).each { CTAnchor anchor -> + if (seenDrawingIds.add(anchor.docPr?.id ?: anchor.xmlText())) { + handler(anchor) + } + } + } + } + } + + private void addBackgroundArea(CTAnchor anchor, DocxPage page) { + if (!isPicture(anchor) || !coversPage(anchor, page)) { + return + } + String embedId = extractBlipEmbedId(anchor) + Position position = new Position(Size.ofPoints(0), Size.ofPoints(0), page.width, page.height) + Area area = embedId ? buildImageArea(anchor, embedId, position) : null + if (area != null) { + backgroundAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + backgroundEmbedIds.add(embedId) + consumedEmbedIds.add(embedId) + } + } + + private void addFloatingArea(CTAnchor anchor, DocxPage page) { + Area area = null + if (anchor.graphic?.graphicData?.uri == WPS_SHAPE_URI) { + area = buildTextBoxArea(anchor, page) + } else if (isPicture(anchor)) { + String embedId = extractBlipEmbedId(anchor) + if (embedId && !backgroundEmbedIds.contains(embedId)) { + area = buildImageArea(anchor, embedId, resolveAnchorPosition(anchor, page)) + if (area != null) { + consumedEmbedIds.add(embedId) + } + } + } + if (area != null) { + floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + } + } + + private static boolean isPicture(CTAnchor anchor) { + return anchor.graphic?.graphicData?.uri == PIC_NS + } + + private static boolean coversPage(CTAnchor anchor, DocxPage page) { + Size width = emuToSize(anchor.extent?.cx ?: 0L) + Size height = emuToSize(anchor.extent?.cy ?: 0L) + return width.toPoints() >= page.width.toPoints() * BACKGROUND_COVERAGE_THRESHOLD + && height.toPoints() >= page.height.toPoints() * BACKGROUND_COVERAGE_THRESHOLD + } + + private static List findAnchors(XWPFRun run) { + XmlObject[] found = run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:anchor") + return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) } + } + + private static String extractBlipEmbedId(CTAnchor anchor) { + XmlObject[] pics = anchor.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic") + if (pics.length == 0) { + return null + } + CTPicture pic = pics[0] instanceof CTPicture ? pics[0] as CTPicture : CTPicture.Factory.parse(pics[0].toString()) + return pic.blipFill?.blip?.embed + } + + private Area buildImageArea(CTAnchor anchor, String embedId, Position position) { + XWPFPictureData data = doc.getPictureDataByID(embedId) + if (data == null) { + return null + } + Size width = emuToSize(anchor.extent?.cx ?: 0L) + Size height = emuToSize(anchor.extent?.cy ?: 0L) + String imageId = registerImageData(migration, data, fileName, new ImageOptions(width, height)) + if (imageId == null) { + return null + } + return new AreaBuilder().imageRef(imageId).position(position).build() + } + + private Area buildTextBoxArea(CTAnchor anchor, DocxPage page) { + XmlCursor cursor = anchor.graphic.graphicData.newCursor() + try { + if (!cursor.toFirstChild()) { + return null + } + def shapeDom = documentBuilder.parse(new InputSource(new StringReader(cursor.xmlText()))) + def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent") + if (txbxContentNodes.length == 0) { + return null + } + List contentItems = [] + def children = txbxContentNodes.item(0).childNodes + for (int i = 0; i < children.length; i++) { + Node child = children.item(i) + if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') { + XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc) + contentItems.add(parseParagraph(migration, paragraph, fileName)) + } + } + if (contentItems.isEmpty()) { + return null + } + String blockId = "${page.id(fileName)}_textbox${++textBoxes}" + String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null } + DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName) + return new AreaBuilder().content([blockRef]).position(resolveAnchorPosition(anchor, page)).build() + } finally { + cursor.dispose() + } + } + + private CTP domParagraphToCtp(Node pNode) { + StringWriter sw = new StringWriter() + Transformer transformer = transformerFactory.newTransformer() + transformer.setOutputProperty(OutputKeys.OMIT_XML_DECLARATION, "yes") + transformer.transform(new DOMSource(pNode), new StreamResult(sw)) + String fragmentXml = sw.toString().replaceFirst(/^$/, '') + return CTP.Factory.parse(fragmentXml) + } + + static Position resolveAnchorPosition(CTAnchor anchor, DocxPage page) { + Size extentWidth = emuToSize(anchor.extent?.cx ?: 0L) + Size extentHeight = emuToSize(anchor.extent?.cy ?: 0L) + double marginLeft = page.contentPosition.x.toPoints() + double marginTop = page.contentPosition.y.toPoints() + boolean relativeToPageH = anchor.positionH?.relativeFrom?.toString() == "page" + boolean relativeToPageV = anchor.positionV?.relativeFrom?.toString() == "page" + + double x + if (anchor.positionH?.isSetPosOffset()) { + x = emuToSize(anchor.positionH.posOffset).toPoints() + (relativeToPageH ? 0.0d : marginLeft) + } else { + double leftBound = relativeToPageH ? 0.0d : marginLeft + double rightBound = relativeToPageH ? page.width.toPoints() : marginLeft + page.contentPosition.width.toPoints() + switch (anchor.positionH?.align?.toString()) { + case "right": + case "outside": + x = rightBound - extentWidth.toPoints() + break + case "center": + x = leftBound + (rightBound - leftBound - extentWidth.toPoints()) / 2.0d + break + default: + x = leftBound + } + } + + double y = anchor.positionV?.isSetPosOffset() + ? emuToSize(anchor.positionV.posOffset).toPoints() + (relativeToPageV ? 0.0d : marginTop) + : marginTop + + return new Position(Size.ofPoints(x), Size.ofPoints(y), extentWidth, extentHeight) + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBlocks.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBlocks.groovy new file mode 100644 index 00000000..3810b759 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBlocks.groovy @@ -0,0 +1,36 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.DisplayRuleRef +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef +import com.quadient.migration.api.dto.migrationmodel.builder.DocumentObjectBuilder +import com.quadient.migration.shared.DocumentObjectType +import groovy.transform.Field + +@Field +static final int MAX_BLOCK_NAME_LENGTH = 60 + +// Stores a piece of content as an internal block and returns the reference to place in the flow instead. +static DocumentObjectRef upsertBlock(Migration migration, String id, String name, List content, String fileName, + String displayRuleId = null) { + migration.documentObjectRepository.upsert(new DocumentObjectBuilder(id, DocumentObjectType.Block) + .name(name) + .content(content) + .internal(true) + .originLocations([fileName]) + .build()) + return new DocumentObjectRef(id, displayRuleId ? new DisplayRuleRef(displayRuleId) : null) +} + +// Derives a readable block name from source text (a heading, a table's first cell, ...), shortened at a word boundary. +static String blockName(String text, String suffix, String fallback) { + String normalized = text?.replaceAll(/\s+/, " ")?.trim() + if (!normalized) { + return fallback + } + if (normalized.length() > MAX_BLOCK_NAME_LENGTH) { + normalized = normalized.substring(0, MAX_BLOCK_NAME_LENGTH).replaceFirst(/\s+\S*$/, "") + } + return suffix ? "${normalized} ${suffix}" : normalized +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy new file mode 100644 index 00000000..9cf22b3d --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy @@ -0,0 +1,125 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.poi.xwpf.usermodel.XWPFTableCell +import org.apache.xmlbeans.XmlObject + +import static com.quadient.migration.example.docx.parser.DocxBlocks.blockName +import static com.quadient.migration.example.docx.parser.DocxBlocks.upsertBlock +import static com.quadient.migration.example.docx.parser.DocxTableParser.addRows +import static com.quadient.migration.example.docx.parser.DocxTableParser.createTableBuilder +import static com.quadient.migration.example.docx.parser.DocxTableParser.haveSameGrid +import static com.quadient.migration.example.docx.parser.DocxTableParser.resolveExpectedColumnCount +import static com.quadient.migration.example.docx.util.DocxUtils.extractParagraphText + +/** + * Collects the body content of one page. The most recent table is kept open (unbuilt) so that a following table wrapped + * in an IF field can be appended to it as conditional rows, which is how Word templates model optional table rows. + * A conditional table that does not fit the preceding one is wrapped in an internal block carrying the display rule. + * Headings are remembered so that the finished page can be divided into one block per top-level section. + */ +class DocxBodyContent { + private final Migration migration + private final String fileName + private final String pageId + private final List content = [] + // Heading level of the content items that are headings, keyed by item identity (model items are value objects). + private final Map headingLevels = new IdentityHashMap<>() + + private TableBuilder openTable + private XWPFTable openTableSource + private int openTableColumnCount + private int conditionalTableBlocks = 0 + + DocxBodyContent(Migration migration, String fileName, String pageId) { + this.migration = migration + this.fileName = fileName + this.pageId = pageId + } + + void add(DocumentContent item, Integer headingLevel = null) { + closeTable() + content.add(item) + if (headingLevel != null) { + headingLevels.put(item, headingLevel) + } + } + + void addTable(XWPFTable table, TableBuilder tableBuilder, int columnCount) { + closeTable() + openTable = tableBuilder + openTableSource = table + openTableColumnCount = columnCount + } + + // Handler for tables wrapped in an IF field. Its rows join a directly preceding regular table with the same grid + // (optional rows of one table); otherwise the whole table is conditional, so it is wrapped in an internal block + // referenced with the display rule, which keeps the table itself free of per-row rules. + void addConditionalTable(XWPFTable table, String displayRuleId) { + int columnCount = resolveExpectedColumnCount(table) + if (openTable != null && openTableColumnCount == columnCount && haveSameGrid(openTableSource, table)) { + addRows(migration, openTable, table, table.rows, fileName, columnCount, false, displayRuleId) + return + } + TableBuilder tableBuilder = createTableBuilder(migration, table, true) + addRows(migration, tableBuilder, table, table.rows, fileName, columnCount, true) + String id = "${pageId}_table${++conditionalTableBlocks}" + add(upsertBlock(migration, id, blockName(firstCellText(table), "table", id), [tableBuilder.build()], fileName, displayRuleId)) + } + + void closeTable() { + if (openTable != null) { + content.add(openTable.build()) + openTable = null + openTableSource = null + } + } + + // The collected content as is, without dividing it into sections. + List flow() { + closeTable() + return content + } + + /** + * Divides the page content into internal blocks, one per top-level heading (the lowest heading level found on the + * page); content before the first heading forms a block of its own. Pages without headings stay a plain flow. + */ + List sectionBlocks() { + closeTable() + Integer topLevel = headingLevels.values().min() + if (topLevel == null) { + return content + } + List> sections = [[]] + content.each { DocumentContent item -> + if (headingLevels[item] == topLevel && !sections.last().isEmpty()) { + sections << [] + } + sections.last() << item + } + return sections.withIndex().collect { List items, int i -> + String id = "${pageId}_section${i + 1}" + String heading = headingLevels[items[0]] == topLevel ? extractParagraphText(items[0]) : null + upsertBlock(migration, id, blockName(heading, null, id), items, fileName) + } + } + + // Text of the first non-empty cell of the table's first row (typically the heading of the row). Word stores the + // text of a table wrapped in a field as instrText, which POI's text accessors skip. + private static String firstCellText(XWPFTable table) { + return table.rows[0].tableCells.collect { cellText(it).trim() }.find { it } + } + + private static String cellText(XWPFTableCell cell) { + return cell.paragraphs.collect { paragraph -> + paragraph.runs.collect { run -> + List fieldChildren = DocxMergeFields.fieldChildren(run) + fieldChildren.isEmpty() ? run.text() : DocxMergeFields.instructionText(fieldChildren) + }.join("") + }.join(" ") + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxFloatingTables.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxFloatingTables.groovy new file mode 100644 index 00000000..cb596a92 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxFloatingTables.groovy @@ -0,0 +1,54 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder +import com.quadient.migration.shared.Position +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTblPPr + +import static com.quadient.migration.example.docx.util.DocxUtils.twipsToPoints + +class DocxFloatingTables { + private static final double MIN_AREA_HEIGHT_PT = 10.0d + + static boolean isFloating(XWPFTable table) { + return table.CTTbl?.tblPr?.tblpPr != null + } + + static Area buildFloatingTableArea(XWPFTable table, List contentItems, Position pagePosition) { + if (contentItems.isEmpty()) { + return null + } + Position position = resolveTablePosition(table, table.CTTbl.tblPr.tblpPr, pagePosition) + return new AreaBuilder().content(contentItems).position(position).build() + } + + private static Position resolveTablePosition(XWPFTable table, CTTblPPr tblpPr, Position pagePosition) { + double marginLeft = pagePosition.x.toPoints() + double marginTop = pagePosition.y.toPoints() + + double x = marginLeft + if (tblpPr.isSetTblpX()) { + x = twipsToPoints(tblpPr.tblpX) + (tblpPr.horzAnchor?.toString() == "page" ? 0.0d : marginLeft) + } + double y = marginTop + if (tblpPr.isSetTblpY()) { + y = twipsToPoints(tblpPr.tblpY) + (tblpPr.vertAnchor?.toString() == "page" ? 0.0d : marginTop) + } + + double contentBottom = marginTop + pagePosition.height.toPoints() + double height = Math.max(contentBottom - y, MIN_AREA_HEIGHT_PT) + return new Position(Size.ofPoints(x), Size.ofPoints(y), resolveTableWidth(table, pagePosition.width), Size.ofPoints(height)) + } + + private static Size resolveTableWidth(XWPFTable table, Size fallbackWidth) { + def gridCols = table.CTTbl.tblGrid?.gridColList + if (!gridCols) { + return fallbackWidth + } + double totalPoints = gridCols.sum { twipsToPoints(it.w) ?: 0.0d } as double + return totalPoints > 0 ? Size.ofPoints(totalPoints) : fallbackWidth + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy new file mode 100644 index 00000000..72523cc3 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy @@ -0,0 +1,281 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.Table +import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder +import com.quadient.migration.shared.ImageOptions +import com.quadient.migration.shared.Position +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.IBodyElement +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFHeaderFooter +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.poi.xwpf.usermodel.XWPFPictureData +import org.apache.xmlbeans.XmlObject +import org.openxmlformats.schemas.drawingml.x2006.picture.CTPicture +import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSectPr + +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseFlowParagraph +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.warnUnterminatedField +import static com.quadient.migration.example.docx.parser.DocxTableParser.parseTable +import static com.quadient.migration.example.docx.parser.DocxImages.registerImageData +import static com.quadient.migration.example.docx.util.DocxUtils.emuToSize +import static com.quadient.migration.example.docx.util.DocxUtils.twipsToPoints + +class DocxHeaderFooters { + private static final List TYPES = ['default', 'first', 'even'] + private static final String WP_NS = 'http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing' + private static final String PIC_NS = 'http://schemas.openxmlformats.org/drawingml/2006/picture' + + static List areas(Migration migration, XWPFDocument doc, DocxPage page, String fileName) { + return headerAreas(migration, doc, page, fileName) + footerAreas(migration, doc, page, fileName) + } + + static List headerAreas(Migration migration, XWPFDocument doc, DocxPage page, String fileName) { + return partAreas(migration, doc, page, fileName, true) + } + + static List footerAreas(Migration migration, XWPFDocument doc, DocxPage page, String fileName) { + return partAreas(migration, doc, page, fileName, false) + } + + private static List partAreas(Migration migration, XWPFDocument doc, DocxPage page, String fileName, boolean headerOnly) { + if (page.sections.isEmpty()) { + return [] + } + Map> parts = effectiveParts(doc, page.sections.first().sectPr) + List result = [] + XWPFHeaderFooter part = selectPart(headerOnly ? parts.headers : parts.footers, page) + if (part != null) { + Position position = headerOnly ? headerPosition(page) : footerPosition(page) + Set directImageEmbedIds = anchoredEmbedIds(part) + inlineEmbedIds(part) + Area area = buildArea(migration, part, fileName, position, directImageEmbedIds) + if (area != null) result.add(area) + result.addAll(inlineImageAreas(migration, doc, part, fileName, position)) + result.addAll(anchoredImageAreas(migration, doc, part, fileName, page, position)) + } + return result + } + + private static Map> effectiveParts(XWPFDocument doc, CTSectPr sectPr) { + Map> result = [headers: [:], footers: [:]] + if (sectPr == null) return result + Map headers = doc.headerList.collectEntries { [(it.packagePart.partName.name): it] } + Map footers = doc.footerList.collectEntries { [(it.packagePart.partName.name): it] } + resolveReferences(doc, sectPr.headerReferenceList, headers, result.headers) + resolveReferences(doc, sectPr.footerReferenceList, footers, result.footers) + return result + } + + private static void resolveReferences(XWPFDocument doc, List references, Map source, + Map target) { + references.each { reference -> + def relationship = doc.packagePart.getRelationship(reference.id) + XWPFHeaderFooter part = relationship == null ? null : source[relationship.targetURI.path] + if (part != null) { + target[reference.type?.toString() ?: 'default'] = part + } + } + } + + private static XWPFHeaderFooter selectPart(Map parts, DocxPage page) { + String type = page.index == 0 && page.sections.first().sectPr?.titlePg != null ? 'first' : + page.index % 2 == 1 ? 'even' : 'default' + // Word falls back to the default part when an even/first-page part is absent. + return parts[type] ?: parts['default'] + } + + private static Area buildArea(Migration migration, XWPFHeaderFooter source, String fileName, Position position, + Set excludedImageEmbedIds = Collections.emptySet()) { + List content = [] + FieldParseState fieldState = new FieldParseState() + source.bodyElements.each { IBodyElement element -> + if (element instanceof XWPFParagraph) { + Paragraph paragraph = parseFlowParagraph(migration, element, fileName, fieldState, 'headerFooter', excludedImageEmbedIds) + if (paragraph?.content) content.add(paragraph) + } else if (element instanceof XWPFTable) { + Table table = parseTable(migration, element, fileName) + if (table != null) content.add(table) + } + } + warnUnterminatedField(fieldState, 'header/footer') + return content.isEmpty() ? null : new AreaBuilder().content(content).position(position).build() + } + + // Anchored drawings are not exposed by XWPFRun.embeddedPictures. Header/footer parts own their image + // relationships, so resolve the blip through that part rather than the document's relationship table. + private static List anchoredImageAreas(Migration migration, XWPFDocument doc, XWPFHeaderFooter source, String fileName, + DocxPage page, Position flowPosition) { + List areas = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + findAnchors(run).each { CTAnchor anchor -> + if (anchor.graphic?.graphicData?.uri != PIC_NS) return + String embedId = extractBlipEmbedId(anchor) + XWPFPictureData data = pictureData(doc, source, embedId) + if (data == null) return + Size width = emuToSize(anchor.extent?.cx ?: 0L) + Size height = emuToSize(anchor.extent?.cy ?: 0L) + String imageId = registerImageData(migration, data, fileName, new ImageOptions(width, height)) + if (imageId != null) { + areas.add(new AreaBuilder().imageRef(imageId) + .position(resolveHeaderFooterAnchorPosition(anchor, page, flowPosition)).build()) + } + } + } + } + return areas + } + + // Header/footer images are fixed page elements. Emit both POI-bound and unbound inline pictures as areas so + // their dimensions are not constrained by the comparatively small header/footer flow frame. + private static List inlineImageAreas(Migration migration, XWPFDocument doc, XWPFHeaderFooter source, String fileName, + Position flowPosition) { + List areas = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + run.embeddedPictures.each { picture -> + CTPicture ctPicture = picture.CTPicture + XWPFPictureData data = picture.pictureData ?: pictureData(doc, source, ctPicture?.blipFill?.blip?.embed) + addInlineImageArea(areas, migration, data, ctPicture, fileName, flowPosition) + } + if (!run.embeddedPictures.isEmpty()) return + findInlines(run).each { XmlObject inline -> + CTPicture picture = inlinePicture(inline) + addInlineImageArea(areas, migration, pictureData(doc, source, picture?.blipFill?.blip?.embed), picture, fileName, flowPosition) + } + } + } + // Last-resort handling for a header/footer part whose inline drawing is not surfaced by POI's run API. + // It is safe only when the document contains exactly one image, which is the package shape of 01CVRPG0222. + if (areas.isEmpty() && doc.allPictures.size() == 1 && source.paragraphs.any { paragraph -> + paragraph.runs.any { it.CTR.xmlText().contains(' findAnchors(XWPFRun run) { + XmlObject[] found = run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:anchor") + return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) } + } + + private static Set anchoredEmbedIds(XWPFHeaderFooter source) { + Set ids = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + findAnchors(run).each { CTAnchor anchor -> + String embedId = extractBlipEmbedId(anchor) + if (embedId) ids.add(embedId) + } + } + } + return ids + } + + private static Set inlineEmbedIds(XWPFHeaderFooter source) { + Set ids = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + run.embeddedPictures.each { picture -> + String embedId = picture.CTPicture?.blipFill?.blip?.embed + if (embedId) ids.add(embedId) + } + findInlines(run).each { XmlObject inline -> + String embedId = inlinePicture(inline)?.blipFill?.blip?.embed + if (embedId) ids.add(embedId) + } + } + } + return ids + } + + private static void addInlineImageArea(List areas, Migration migration, XWPFPictureData data, CTPicture picture, + String fileName, Position flowPosition) { + if (data == null || picture == null) return + Size width = emuToSize(picture.spPr?.xfrm?.ext?.cx ?: 0L) + Size height = emuToSize(picture.spPr?.xfrm?.ext?.cy ?: 0L) + String imageId = registerImageData(migration, data, fileName, new ImageOptions(width, height)) + if (imageId != null) { + areas.add(new AreaBuilder().imageRef(imageId) + .position(new Position(flowPosition.x, flowPosition.y, width, height)).build()) + } + } + + private static List findInlines(XWPFRun run) { + return run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:inline") as List + } + + private static CTPicture inlinePicture(XmlObject inline) { + XmlObject[] pictures = inline.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic") + return pictures.length == 0 ? null : (pictures[0] instanceof CTPicture + ? pictures[0] as CTPicture : CTPicture.Factory.parse(pictures[0].xmlText())) + } + + // Header/footer anchors can be relative to their paragraph or column. Unlike body content, those origins sit + // in the header/footer frame, not at the main body margin. + private static Position resolveHeaderFooterAnchorPosition(CTAnchor anchor, DocxPage page, Position flowPosition) { + Size width = emuToSize(anchor.extent?.cx ?: 0L) + Size height = emuToSize(anchor.extent?.cy ?: 0L) + boolean relativeToPageH = anchor.positionH?.relativeFrom?.toString() == 'page' + boolean relativeToPageV = anchor.positionV?.relativeFrom?.toString() == 'page' + double x = relativeToPageH ? 0d : flowPosition.x.toPoints() + double y = relativeToPageV ? 0d : flowPosition.y.toPoints() + if (anchor.positionH?.isSetPosOffset()) x += emuToSize(anchor.positionH.posOffset).toPoints() + if (anchor.positionV?.isSetPosOffset()) y += emuToSize(anchor.positionV.posOffset).toPoints() + return new Position(Size.ofPoints(x), Size.ofPoints(y), width, height) + } + + private static XWPFPictureData pictureData(XWPFDocument doc, XWPFHeaderFooter source, String embedId) { + if (!embedId) return null + XWPFPictureData data = source.getPictureDataByID(embedId) + if (data != null) return data + def relatedPart = source.getRelationById(embedId) + if (relatedPart instanceof XWPFPictureData) return relatedPart + def relationship = source.packagePart.getRelationship(embedId) + data = relationship == null ? null : source.allPictures.find { + it.packagePart.partName.name == relationship.targetURI.path + } + // Some DOCX producers leave POI unable to bind a header relationship even though the document has a + // single embedded image. This fallback covers that unambiguous package shape (including 01CVRPG0222). + return data ?: (doc.allPictures.size() == 1 ? doc.allPictures[0] : null) + } + + private static String extractBlipEmbedId(CTAnchor anchor) { + XmlObject[] pictures = anchor.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic") + if (pictures.length == 0) return null + CTPicture picture = pictures[0] instanceof CTPicture ? pictures[0] as CTPicture : CTPicture.Factory.parse(pictures[0].xmlText()) + return picture.blipFill?.blip?.embed + } + + private static Position headerPosition(DocxPage page) { + double contentTop = page.contentPosition.y.toPoints() + double headerOffset = headerFooterOffset(page, 'header', 0d) + double y = Math.min(Math.max(headerOffset, 0d), contentTop) + return new Position(page.contentPosition.x, Size.ofPoints(y), page.contentPosition.width, + Size.ofPoints(Math.max(contentTop - y, 1d))) + } + + private static Position footerPosition(DocxPage page) { + double y = page.contentPosition.y.toPoints() + page.contentPosition.height.toPoints() + double height = page.height.toPoints() - y + return new Position(page.contentPosition.x, Size.ofPoints(y), page.contentPosition.width, Size.ofPoints(Math.max(height, 1d))) + } + + private static double headerFooterOffset(DocxPage page, String property, double fallback) { + def margins = page.sections.first()?.sectPr?.pgMar + Double offset = twipsToPoints(margins?."${property}") + return offset != null ? offset : fallback + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeadings.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeadings.groovy new file mode 100644 index 00000000..ac8c041f --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeadings.groovy @@ -0,0 +1,65 @@ +package com.quadient.migration.example.docx.parser + +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFStyle +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTStyle + +import java.util.regex.Matcher +import java.util.regex.Pattern + +// Word's outline levels are 0-based and level 9 means "body text". +@Field +static final int BODY_TEXT_OUTLINE_LEVEL = 9 +@Field +static final Pattern HEADING_STYLE_NAME = ~/(?i)(heading|hdr|title)\s*(\d)?/ +// Styles whose name says they are body text are never headings, even when they are (carelessly) based on a heading. +@Field +static final Pattern BODY_STYLE_NAME = ~/(?i)body|normal|text|list|table|toc|caption|footer|header/ + +/** + * Returns the 1-based heading level of a paragraph, or null when it is ordinary text. The paragraph's style chain is + * walked from the most specific style upwards: a heading-like style name decides first (e.g. "SLR HDR2" or "1. SLR + * HDR1", which have no outline level of their own), then a style's outline level. Outline levels set directly on a + * paragraph are ignored, as they typically stem from copied formatting rather than from document structure. + */ +static Integer headingLevel(XWPFParagraph paragraph) { + if (paragraph.text.isBlank()) { + return null + } + Set visited = [] + String styleId = paragraph.styleID + while (styleId && visited.add(styleId)) { + XWPFStyle style = paragraph.document.styles?.getStyle(styleId) + if (style == null) { + return null + } + Integer level = levelOf(style) + if (level != null) { + return level > 0 ? level : null + } + CTStyle ctStyle = style.CTStyle + styleId = ctStyle.isSetBasedOn() ? ctStyle.basedOn.val : null + } + return null +} + +// Heading level of one style; 0 means "definitely body text", null means "undecided, ask the parent style". +private static Integer levelOf(XWPFStyle style) { + String name = style.name ?: style.styleId + if (name =~ /(?i)subtitle/) { + return 2 + } + Integer outlineLevel = style.CTStyle.PPr?.isSetOutlineLvl() ? style.CTStyle.PPr.outlineLvl.val.intValue() : null + if (outlineLevel != null && outlineLevel >= BODY_TEXT_OUTLINE_LEVEL) { + return 0 + } + Matcher heading = HEADING_STYLE_NAME.matcher(name) + if (heading.find()) { + return heading.group(2) ? heading.group(2).toInteger() : (outlineLevel != null ? outlineLevel + 1 : 1) + } + if (BODY_STYLE_NAME.matcher(name).find()) { + return 0 + } + return outlineLevel != null ? outlineLevel + 1 : null +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy new file mode 100644 index 00000000..6e8ea0cb --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy @@ -0,0 +1,115 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.builder.ImageBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import com.quadient.migration.shared.ImageOptions +import com.quadient.migration.shared.ImageType +import com.quadient.migration.shared.Size +import groovy.transform.Field +import org.apache.poi.common.usermodel.PictureType +import org.apache.poi.xwpf.usermodel.XWPFPicture +import org.apache.poi.xwpf.usermodel.XWPFPictureData +import org.apache.poi.xwpf.usermodel.XWPFRun + +@Field +static Map imageIdByChecksum = [:] +@Field +static int imageCounter = 0 + +static void resetImageState() { + imageIdByChecksum = [:] + imageCounter = 0 +} + +static void processRunImages(Migration migration, XWPFRun run, String fileName, List textBuilders, Set excludedEmbedIds = Collections.emptySet()) { + run.getEmbeddedPictures().each { XWPFPicture picture -> + XWPFPictureData data = picture.getPictureData() + if (data == null) { + return + } + String embedId = picture.getCTPicture()?.getBlipFill()?.getBlip()?.getEmbed() + if (embedId && excludedEmbedIds.contains(embedId)) { + return + } + String imageId = registerImageData(migration, data, fileName, resolveOptions(picture)) + if (imageId == null) { + return + } + ParagraphBuilder.TextBuilder textBuilder = new ParagraphBuilder.TextBuilder() + textBuilder.imageRef(imageId) + textBuilders.add(textBuilder) + } +} + +static String registerImageData(Migration migration, XWPFPictureData data, String fileName, ImageOptions options) { + Long checksum = data.getChecksum() + if (checksum != null && imageIdByChecksum.containsKey(checksum)) { + return imageIdByChecksum[checksum] + } + + ImageType imageType = toImageType(data.getPictureTypeEnum()) + if (imageType == ImageType.Unknown) { + // Unsupported/unrecognized format (e.g. EMF/WMF vector metafiles) - skip rather than upsert unusable data. + println " Warning: Skipping embedded image with unsupported type: ${data.getPictureTypeEnum()}" + return null + } + + String imageId = "${fileName}_img_${++imageCounter}" + String storagePath = "${imageId}${imageType.extension()}" + + // Storage has both String and byte[] overloads. Keep the declared type so a malformed/empty picture payload + // cannot make Groovy select neither overload at runtime. + byte[] imageBytes = data.getData() + migration.storage.write(storagePath, imageBytes) + + def imageBuilder = new ImageBuilder(imageId) + .sourcePath(storagePath) + .imageType(imageType) + .originLocations([fileName]) + + if (options) { + imageBuilder.options(options) + } + + migration.imageRepository.upsert(imageBuilder.build()) + + if (checksum != null) { + imageIdByChecksum[checksum] = imageId + } + return imageId +} + +private static ImageOptions resolveOptions(XWPFPicture picture) { + try { + double widthPt = picture.getWidth() + double heightPt = picture.getDepth() + if (widthPt > 0 && heightPt > 0) { + return new ImageOptions(Size.ofPoints(widthPt), Size.ofPoints(heightPt)) + } + } catch (Exception ignored) { + // Picture is missing shape xfrm/ext data - skip sizing rather than fail the whole run. + } + return null +} + +private static ImageType toImageType(PictureType pictureType) { + switch (pictureType) { + case PictureType.PNG: + return ImageType.Png + case PictureType.JPEG: + case PictureType.CMYKJPEG: + return ImageType.Jpeg + case PictureType.GIF: + return ImageType.Gif + case PictureType.BMP: + case PictureType.DIB: + return ImageType.Bmp + case PictureType.TIFF: + return ImageType.Tiff + case PictureType.SVG: + return ImageType.Svg + default: + return ImageType.Unknown + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy new file mode 100644 index 00000000..87a94da9 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy @@ -0,0 +1,585 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.DisplayRuleRef +import com.quadient.migration.api.dto.migrationmodel.FirstMatch +import com.quadient.migration.api.dto.migrationmodel.StringValue +import com.quadient.migration.api.dto.migrationmodel.builder.DisplayRuleBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.VariableBuilder +import com.quadient.migration.shared.BinOp +import com.quadient.migration.shared.Binary +import com.quadient.migration.shared.DataType +import com.quadient.migration.shared.DisplayRuleDefinition +import com.quadient.migration.shared.Group +import com.quadient.migration.shared.GroupOp +import com.quadient.migration.shared.Literal +import com.quadient.migration.shared.LiteralDataType +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.xmlbeans.XmlObject +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTFldChar +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTText + +import static com.quadient.migration.example.docx.util.DocxUtils.sha256Hex + +interface FieldPart {} + +class FieldTextPart implements FieldPart { + String text + String styleId +} + +// A body-level table that Word placed inside a field instruction (a whole table wrapped in an IF field). +class FieldTablePart implements FieldPart { + XWPFTable table +} + +class WordField implements FieldPart { + List instructionParts = [] + String styleId + boolean resultStarted = false + + void addPart(FieldPart part) { + // Only the instruction is interpreted; the cached field result (after "separate") is ignored. + if (!resultStarted) { + instructionParts.add(part) + } + } + + String instructionText() { + return instructionParts.findAll { it instanceof FieldTextPart } + .collect { (it as FieldTextPart).text } + .join("") + } + + boolean containsTable() { + return instructionParts.any { it instanceof FieldTablePart || (it instanceof WordField && it.containsTable()) } + } +} + +class FieldParseState { + List stack = [] + List pendingIfFields = [] + boolean pendingFirstMatchEligible = true + // Receives (XWPFTable, displayRuleId) for tables wrapped in a resolved IF field; null when tables cannot occur. + Closure conditionalTableHandler + + int getDepth() { + return stack.size() + } + + boolean isInsideFieldResult() { + return stack.any { it.resultStarted } + } +} + +class IfCondition { + String variableId + String operator + String rightOperand + boolean negated + + String canonical() { + return "${variableId} ${effectiveOperator()} ${rightOperand}" + } + + // A negated (else-branch) condition is expressed by inverting the operator so the rule stays a flat AND group. + String effectiveOperator() { + if (!negated) { + return operator + } + return switch (operator) { + case "=" -> "<>" + case "<>" -> "=" + case ">" -> "<=" + case "<=" -> ">" + case "<" -> ">=" + case ">=" -> "<" + default -> throw new IllegalArgumentException("Unsupported Word IF operator: ${operator}") + } + } + + Literal operandLiteral() { + if (rightOperand.startsWith('"')) { + return new Literal(rightOperand.substring(1, rightOperand.length() - 1), LiteralDataType.String) + } + return new Literal(rightOperand, LiteralDataType.Number) + } +} + +class ParsedIfField { + String variableId + String operator + String rightOperand + // Branch content is text, nested fields (IF / MERGEFIELD) or tables, in source order. + List trueContent + List falseContent + + String condition() { + return "${variableId} ${operator} ${rightOperand}" + } + + IfCondition toCondition(boolean negated) { + return new IfCondition(variableId: variableId, operator: operator, rightOperand: rightOperand, negated: negated) + } + + Literal operandLiteral() { + return toCondition(false).operandLiteral() + } +} + +@Field +static final char PLACEHOLDER = (char) 0xE000 + +static List fieldChildren(XWPFRun run) { + return run.CTR.selectPath("./*").findAll { it.domNode.localName in ["fldChar", "instrText"] } +} + +static boolean isFieldBegin(XmlObject child) { + return child.domNode.localName == "fldChar" && (child as CTFldChar).fldCharType.toString() == "begin" +} + +static boolean isInstructionText(XmlObject child) { + return child.domNode.localName == "instrText" +} + +static String instructionText(List fieldChildren) { + return fieldChildren.findAll { isInstructionText(it) }.collect { (it as CTText).stringValue }.join("") +} + +static void handleFieldChildren(Migration migration, List fieldChildren, String textStyleId, FieldParseState state, + List textBuilders, String fileName) { + fieldChildren.each { XmlObject child -> + if (child.domNode.localName == "fldChar") { + String type = (child as CTFldChar).fldCharType.toString() + handleFieldCharacter(migration, type, textStyleId, state, textBuilders, fileName) + } else if (state.depth > 0) { + state.stack.last().addPart(new FieldTextPart(text: (child as CTText).stringValue, styleId: textStyleId)) + } + } +} + +// Called for a body-level table encountered while a field is open; returns false when the table is a cached field result +// (content between "separate" and "end") that must not be emitted at all. +static boolean handleTableInsideField(FieldParseState state, XWPFTable table) { + if (state.insideFieldResult) { + return false + } + state.stack.last().addPart(new FieldTablePart(table: table)) + return true +} + +private static void handleFieldCharacter(Migration migration, String type, String textStyleId, FieldParseState state, + List textBuilders, String fileName) { + switch (type) { + case "begin": + WordField field = new WordField(styleId: textStyleId) + if (state.depth > 0) { + state.stack.last().addPart(field) + } + state.stack.add(field) + break + case "separate": + if (state.depth > 0) { + state.stack.last().resultStarted = true + } + break + case "end": + if (state.depth == 0) { + return + } + WordField completed = state.stack.removeLast() + if (state.depth == 0) { + resolveField(migration, state, textBuilders, fileName, completed) + } + break + } +} + +private static void resolveField(Migration migration, FieldParseState state, List textBuilders, String fileName, + WordField field) { + String instruction = field.instructionText() + if (isMergeField(instruction)) { + // A MERGEFIELD (or an unsupported field below) breaks a chain of IFs. + flushPendingIfFields(migration, state, textBuilders, fileName) + addMergeField(migration, textBuilders, fileName, instruction, field.styleId) + return + } + + ParsedIfField parsed = parseIfField(field) + if (parsed == null) { + flushPendingIfFields(migration, state, textBuilders, fileName) + if (field.containsTable()) { + println " Warning: Unsupported field wrapping a table, its tables are dropped: '${instruction.trim()}'" + } + return + } + + ensureVariable(migration, parsed.variableId, fileName) + if (!state.pendingIfFields.isEmpty() && state.pendingIfFields.first().variableId != parsed.variableId) { + flushPendingIfFields(migration, state, textBuilders, fileName) + } + state.pendingFirstMatchEligible = state.pendingIfFields.isEmpty() + ? isFirstMatchCandidate(parsed) + : state.pendingFirstMatchEligible && canJoinFirstMatch(state.pendingIfFields, parsed) + state.pendingIfFields.add(parsed) +} + +static void flushPendingIfFields(Migration migration, FieldParseState state, List textBuilders, String fileName) { + if (state.pendingIfFields.isEmpty()) { + return + } + + if (state.pendingIfFields.size() > 1 && state.pendingFirstMatchEligible) { + addFirstMatch(migration, textBuilders, fileName, state.pendingIfFields) + } else { + state.pendingIfFields.each { addIndependentConditional(migration, state, textBuilders, fileName, it) } + } + state.pendingIfFields.clear() + state.pendingFirstMatchEligible = true +} + +private static void addIndependentConditional(Migration migration, FieldParseState state, List textBuilders, + String fileName, ParsedIfField parsed) { + boolean blockLevel = containsTable(parsed.trueContent) || containsTable(parsed.falseContent) + emitBranch(migration, state, textBuilders, fileName, parsed.trueContent, [parsed.toCondition(false)], blockLevel) + emitBranch(migration, state, textBuilders, fileName, parsed.falseContent, [parsed.toCondition(true)], blockLevel) +} + +// Emits one IF branch; every nested IF extends the condition path, so each leaf gets a rule that ANDs all conditions +// leading to it. Whitespace-only text is dropped for fields wrapping tables, where it is just field-code formatting. +private static void emitBranch(Migration migration, FieldParseState state, List textBuilders, String fileName, + List parts, List path, boolean blockLevel) { + parts.each { FieldPart part -> + if (part instanceof FieldTextPart) { + if (part.text && !(blockLevel && part.text.isBlank())) { + textBuilders.add(new ParagraphBuilder.TextBuilder() + .string(part.text) + .styleRef(part.styleId) + .displayRuleRef(upsertDisplayRule(migration, path, fileName))) + } + } else if (part instanceof FieldTablePart) { + if (state.conditionalTableHandler != null) { + state.conditionalTableHandler.call(part.table, upsertDisplayRule(migration, path, fileName)) + } else { + println " Warning: Table inside IF field is not supported in this context and is dropped." + } + } else if (part instanceof WordField) { + emitNestedField(migration, state, textBuilders, fileName, part, path, blockLevel) + } + } +} + +private static void emitNestedField(Migration migration, FieldParseState state, List textBuilders, String fileName, + WordField field, List path, boolean blockLevel) { + String instruction = field.instructionText() + if (isMergeField(instruction)) { + String variableId = extractMergeFieldName(instruction) + if (variableId) { + ensureVariable(migration, variableId, fileName) + textBuilders.add(new ParagraphBuilder.TextBuilder() + .variableRef(variableId) + .styleRef(field.styleId) + .displayRuleRef(upsertDisplayRule(migration, path, fileName))) + } + return + } + ParsedIfField nested = parseIfField(field) + if (nested == null) { + println " Warning: Unsupported nested field inside IF branch is dropped: '${instruction.trim()}'" + return + } + ensureVariable(migration, nested.variableId, fileName) + emitBranch(migration, state, textBuilders, fileName, nested.trueContent, path + [nested.toCondition(false)], blockLevel) + emitBranch(migration, state, textBuilders, fileName, nested.falseContent, path + [nested.toCondition(true)], blockLevel) +} + +private static boolean containsTable(List parts) { + return parts.any { it instanceof FieldTablePart || (it instanceof WordField && it.containsTable()) } +} + +private static void addFirstMatch(Migration migration, List textBuilders, String fileName, + List fields) { + List cases = fields.collect { field -> + String ruleId = upsertDisplayRule(migration, [field.toCondition(false)], fileName) + new FirstMatch.Case( + new DisplayRuleRef(ruleId), + [new StringValue(field.trueContent*.text.join(""))], + field.condition() + ) + } + textBuilders.add(new ParagraphBuilder.TextBuilder() + .styleRef((fields.first().trueContent.first() as FieldTextPart).styleId) + .firstMatch(new FirstMatch(cases, []))) +} + +private static boolean canJoinFirstMatch(List fields, ParsedIfField candidate) { + return isFirstMatchCandidate(candidate) + && fields.every { isFirstMatchCandidate(it) } + && fields.first().variableId == candidate.variableId + && (fields.first().trueContent.first() as FieldTextPart).styleId == (candidate.trueContent.first() as FieldTextPart).styleId + && fields.every { it.operandLiteral() != candidate.operandLiteral() } +} + +private static boolean isFirstMatchCandidate(ParsedIfField field) { + return field.operator == "=" + && field.falseContent.isEmpty() + && field.trueContent.size() == 1 + && field.trueContent.first() instanceof FieldTextPart + && (field.trueContent.first() as FieldTextPart).text +} + +static void addMergeField(Migration migration, List textBuilders, String fileName, + String fieldInstruction, String textStyleId) { + String variableId = extractMergeFieldName(fieldInstruction) + if (!variableId) { + return + } + ensureVariable(migration, variableId, fileName) + textBuilders.add(new ParagraphBuilder.TextBuilder() + .variableRef(variableId) + .styleRef(textStyleId)) +} + +private static void ensureVariable(Migration migration, String variableId, String fileName) { + if (migration.variableRepository.find(variableId) == null) { + migration.variableRepository.upsert(new VariableBuilder(variableId) + .name(variableId) + .originLocations([fileName]) + .dataType(DataType.String) + .build()) + } +} + +static String upsertDisplayRule(Migration migration, List conditions, String fileName) { + String canonical = conditions*.canonical().join(" AND ") + String ruleId = "docx_if_${sha256Hex(canonical).substring(0, 16)}" + if (migration.displayRuleRepository.find(ruleId) == null) { + migration.displayRuleRepository.upsert(new DisplayRuleBuilder(ruleId) + .name(canonical) + .originLocations([fileName]) + .addCustomField("originContent", canonical) + .definition(parseConditions(conditions)) + .build()) + } + return ruleId +} + +static DisplayRuleDefinition parseConditions(List conditions) { + List comparisons = conditions.collect { IfCondition condition -> + BinOp operator = switch (condition.effectiveOperator()) { + case "=" -> BinOp.Equals + case "<>" -> BinOp.NotEquals + case ">" -> BinOp.GreaterThan + case "<" -> BinOp.LessThan + case ">=" -> BinOp.GreaterOrEqualThan + case "<=" -> BinOp.LessOrEqualThen + default -> throw new IllegalArgumentException("Unsupported Word IF operator: ${condition.operator}") + } + new Binary(new Literal(condition.variableId, LiteralDataType.Variable), operator, condition.operandLiteral()) + } + return new DisplayRuleDefinition(new Group(comparisons, GroupOp.And, false)) +} + +class InstructionSpan { + int start + int end + FieldPart part +} + +class Token { + int contentStart + int contentEnd + int end +} + +static ParsedIfField parseIfField(WordField field) { + List spans = [] + StringBuilder flattened = new StringBuilder() + + // Nested fields and tables are replaced by private-use placeholder chars so the instruction can be scanned as one string. + field.instructionParts.each { FieldPart part -> + int start = flattened.length() + if (part instanceof FieldTextPart) { + flattened.append(part.text) + } else { + flattened.append(PLACEHOLDER) + } + spans.add(new InstructionSpan(start: start, end: flattened.length(), part: part)) + } + + String input = flattened.toString() + int index = skipWhitespace(input, 0) + if (!startsWithWordIgnoreCase(input, index, "IF")) { + return null + } + index = skipWhitespace(input, index + 2) + if (index >= input.length() || input.charAt(index) != PLACEHOLDER) { + return null + } + + FieldPart left = spans.find { it.start == index && !(it.part instanceof FieldTextPart) }?.part + if (!(left instanceof WordField)) { + return null + } + String variableId = extractMergeFieldName((left as WordField).instructionText()) + if (!variableId) { + return null + } + + index = skipWhitespace(input, index + 1) + String operator = [">=", "<=", "<>", "=", ">", "<"].find { input.startsWith(it, index) } + if (!operator) { + return null + } + index = skipWhitespace(input, index + operator.length()) + + Token right = readOperand(input, index) + if (right == null) { + return null + } + String rightOperand = input.substring(right.contentStart, right.end) + if (!isSupportedOperand(rightOperand)) { + return null + } + + Token trueBranch = readBranch(input, skipWhitespace(input, right.end)) + if (trueBranch == null) { + return null + } + // Word ignores anything after the false branch (switches, stray quotes), so no trailing validation is done. + Token falseBranch = readBranch(input, skipWhitespace(input, trueBranch.end)) + + return new ParsedIfField( + variableId: variableId, + operator: operator, + rightOperand: rightOperand, + trueContent: extractRange(spans, trueBranch), + falseContent: falseBranch == null ? [] : extractRange(spans, falseBranch) + ) +} + +private static boolean isSupportedOperand(String operand) { + return (operand.length() >= 2 && operand.startsWith('"') && operand.endsWith('"')) + || operand ==~ /[-+]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][-+]?\d+)?/ +} + +// The returned token spans the whole operand including quotes, so contentStart is the operand start. +private static Token readOperand(String input, int start) { + if (start >= input.length()) { + return null + } + if (input.charAt(start) == '"') { + Token quoted = readQuoted(input, start) + return quoted == null ? null : new Token(contentStart: start, contentEnd: quoted.end, end: quoted.end) + } + return readBareToken(input, start) +} + +// A branch is either a quoted string or a single bare token (typically just a nested field placeholder). +private static Token readBranch(String input, int start) { + if (start >= input.length()) { + return null + } + if (input.charAt(start) == '"') { + return readQuoted(input, start) + } + if (input.charAt(start) == '\\') { + return null + } + return readBareToken(input, start) +} + +private static Token readBareToken(String input, int start) { + int end = start + while (end < input.length() && !Character.isWhitespace(input.charAt(end))) { + end++ + } + return end == start ? null : new Token(contentStart: start, contentEnd: end, end: end) +} + +// Like Word, an unterminated quote runs to the end of the instruction. +private static Token readQuoted(String input, int start) { + if (start >= input.length() || input.charAt(start) != '"') { + return null + } + int index = start + 1 + while (index < input.length()) { + if (input.charAt(index) == '"' && (index == start + 1 || input.charAt(index - 1) != '\\')) { + return new Token(contentStart: start + 1, contentEnd: index, end: index + 1) + } + index++ + } + return new Token(contentStart: start + 1, contentEnd: input.length(), end: input.length()) +} + +private static List extractRange(List spans, Token range) { + List result = [] + spans.each { InstructionSpan span -> + int overlapStart = Math.max(range.contentStart, span.start) + int overlapEnd = Math.min(range.contentEnd, span.end) + if (overlapStart >= overlapEnd) { + return + } + if (!(span.part instanceof FieldTextPart)) { + result.add(span.part) + return + } + FieldTextPart textPart = span.part as FieldTextPart + String text = textPart.text.substring(overlapStart - span.start, overlapEnd - span.start) + FieldTextPart previous = result && result.last() instanceof FieldTextPart ? result.last() as FieldTextPart : null + // Whitespace carries no visible formatting, so it is glued to the neighbouring text instead of becoming a + // separate (ruled) text of its own. + if (previous != null && (previous.styleId == textPart.styleId || text.isBlank())) { + previous.text += text + } else if (previous != null && previous.text.isBlank()) { + previous.text += text + previous.styleId = textPart.styleId + } else { + result.add(new FieldTextPart(text: text, styleId: textPart.styleId)) + } + } + return result +} + +private static int skipWhitespace(String input, int index) { + while (index < input.length() && Character.isWhitespace(input.charAt(index))) { + index++ + } + return index +} + +// A nested field directly after the keyword (e.g. "if«MERGEFIELD X»") also terminates the word, as in Word. +private static boolean startsWithWordIgnoreCase(String input, int index, String word) { + if (index + word.length() > input.length() + || !input.regionMatches(true, index, word, 0, word.length())) { + return false + } + int end = index + word.length() + return end == input.length() || Character.isWhitespace(input.charAt(end)) || input.charAt(end) == PLACEHOLDER +} + +static boolean isMergeField(String fieldInstruction) { + return fieldInstruction?.trim()?.toUpperCase(Locale.ROOT)?.startsWith("MERGEFIELD") +} + +static String extractMergeFieldName(String fieldInstruction) { + String instr = fieldInstruction?.trim() + if (!instr) { + return null + } + String remainder = instr.replaceFirst(/(?i)^MERGEFIELD\b\s*/, "").trim() + if (!remainder) { + return null + } + def quoted = (remainder =~ /^"([^"]+)"/) + if (quoted.find()) { + return quoted.group(1).trim() + } + def switchMatcher = (remainder =~ /\\\S/) + String name = switchMatcher.find() ? remainder.substring(0, switchMatcher.start()) : remainder + name = name.trim() + return name ?: null +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy new file mode 100644 index 00000000..2854d2f7 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy @@ -0,0 +1,52 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.shared.Position +import com.quadient.migration.shared.Size +import groovy.transform.Field +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPageMar +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSectPr + +import static com.quadient.migration.example.docx.util.DocxUtils.twipsToPoints + +@Field +static final List DEFAULT_PAGE_SIZE = [Size.ofMillimeters(210), Size.ofMillimeters(297)] +@Field +static final Map DEFAULT_MARGINS = [ + left: Size.ofMillimeters(20), right: Size.ofMillimeters(20), + top: Size.ofMillimeters(20), bottom: Size.ofMillimeters(37), +] +// Preserve the existing 170 x 240 mm fallback content area on an A4 page. +@Field +static final Position DEFAULT_CONTENT_POSITION = new Position( + DEFAULT_MARGINS.left, DEFAULT_MARGINS.top, + Size.ofPoints(DEFAULT_PAGE_SIZE[0].toPoints() - DEFAULT_MARGINS.left.toPoints() - DEFAULT_MARGINS.right.toPoints()), + Size.ofPoints(DEFAULT_PAGE_SIZE[1].toPoints() - DEFAULT_MARGINS.top.toPoints() - DEFAULT_MARGINS.bottom.toPoints())) + +static Position resolvePageContentPosition(CTSectPr sectPr) { + CTPageMar pageMar = sectPr?.pgMar + def (pageWidth, pageHeight) = resolvePageSize(sectPr)*.toPoints() + Size top = resolveDimension(pageMar?.top, DEFAULT_MARGINS.top) + Size left = resolveDimension(pageMar?.left, DEFAULT_MARGINS.left) + double marginTop = top.toPoints() + double marginRight = resolveDimension(pageMar?.right, DEFAULT_MARGINS.right).toPoints() + double marginBottom = resolveDimension(pageMar?.bottom, DEFAULT_MARGINS.bottom).toPoints() + double marginLeft = left.toPoints() + + double contentWidth = pageWidth - marginLeft - marginRight + double contentHeight = pageHeight - marginTop - marginBottom + if (contentWidth <= 0 || contentHeight <= 0) { + return DEFAULT_CONTENT_POSITION + } + return new Position(left, top, Size.ofPoints(contentWidth), Size.ofPoints(contentHeight)) +} + +static List resolvePageSize(CTSectPr sectPr) { + return [resolveDimension(sectPr?.pgSz?.w, DEFAULT_PAGE_SIZE[0]), + resolveDimension(sectPr?.pgSz?.h, DEFAULT_PAGE_SIZE[1])] +} + +private static Size resolveDimension(Object twips, Size fallback) { + Double points = twipsToPoints(twips) + // Zero is a valid explicit margin, so do not use Groovy's truth-based fallback. + return points != null ? Size.ofPoints(points) : fallback +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy new file mode 100644 index 00000000..cb0a989b --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy @@ -0,0 +1,124 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.shared.Position +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.IBodyElement +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSectPr +import org.openxmlformats.schemas.wordprocessingml.x2006.main.STSectionMark + +class DocxSection { + List elements = [] + CTSectPr sectPr +} + +class DocxPage { + int index + List sections = [] + Position contentPosition + Size width + Size height + List content = [] + + List bodyElements() { + return sections.collectMany { it.elements } + } + + String id(String fileName) { + return "${fileName}_page${index + 1}" + } + + List paragraphs() { + return bodyElements().findAll { it instanceof XWPFParagraph } as List + } +} + +class DocxPageSections { + static List splitIntoSections(XWPFDocument doc) { + List sections = [] + DocxSection current = new DocxSection() + doc.bodyElements.each { IBodyElement elem -> + current.elements.add(elem) + CTSectPr sectPr = elem instanceof XWPFParagraph ? elem.CTP.PPr?.sectPr : null + if (sectPr != null) { + current.sectPr = sectPr + sections.add(current) + current = new DocxSection() + } + } + if (!current.elements.isEmpty()) { + current.sectPr = doc.document?.body?.sectPr + sections.add(current) + } + return sections + } + + static List groupIntoPages(List sections) { + List pages = [] + sections.eachWithIndex { DocxSection section, int i -> + // A section can contain Word's cached pagination marker. It is not an authored page break, so only + // honor it where it begins a paragraph; using a marker after text would move already-laid-out content. + splitAtRenderedPageBreaks(section).eachWithIndex { DocxSection fragment, int fragmentIndex -> + if ((i == 0 && fragmentIndex == 0) || fragmentIndex > 0 || startsNewPage(section.sectPr)) { + pages.add(newPage(pages.size(), fragment)) + } else { + pages.last().sections.add(fragment) + } + } + } + return pages + } + + private static List splitAtRenderedPageBreaks(DocxSection section) { + if (!section.elements.any { it instanceof XWPFParagraph && beginsAfterRenderedPageBreak(it) }) { + // Preserve the original section object when no cached pagination is involved. + return [section] + } + List fragments = [] + DocxSection current = new DocxSection(sectPr: section.sectPr) + section.elements.each { IBodyElement element -> + if (element instanceof XWPFParagraph && beginsAfterRenderedPageBreak(element) && !current.elements.isEmpty()) { + fragments.add(current) + current = new DocxSection(sectPr: section.sectPr) + } + current.elements.add(element) + } + if (!current.elements.isEmpty()) { + fragments.add(current) + } + return fragments + } + + private static boolean beginsAfterRenderedPageBreak(XWPFParagraph paragraph) { + String xml = paragraph.CTP.xmlText() + def marker = xml =~ /<(?:[A-Za-z_][\w.-]*:)?lastRenderedPageBreak(?=[\s\/>])/ + if (!marker.find()) { + return false + } + // These are the WordprocessingML elements that produce visible paragraph content. The marker is safe to + // treat as a page boundary only when none of them precedes it. + def visibleContent = xml =~ /<(?:[A-Za-z_][\w.-]*:)?(?:t|instrText|drawing|tab|br|object)(?=[\s\/>])/ + return !visibleContent.find() || marker.start() < visibleContent.start() + } + + private static DocxPage newPage(int index, DocxSection firstSection) { + List size = DocxPageLayout.resolvePageSize(firstSection.sectPr) + return new DocxPage( + index: index, + sections: [firstSection], + contentPosition: DocxPageLayout.resolvePageContentPosition(firstSection.sectPr), + width: size[0], + height: size[1], + ) + } + + private static boolean startsNewPage(CTSectPr sectPr) { + if (sectPr == null || !sectPr.isSetType()) { + return true + } + def mark = sectPr.type.val + return mark != STSectionMark.CONTINUOUS && mark != STSectionMark.NEXT_COLUMN + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy new file mode 100644 index 00000000..7d759457 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy @@ -0,0 +1,150 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFFieldRun +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.xmlbeans.XmlObject +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSimpleField + +import static com.quadient.migration.example.docx.style.DocxParagraphStyles.captureParagraphStyle +import static com.quadient.migration.example.docx.style.DocxTextStyles.captureTextStyle + +class ParagraphContentCollector { + private final Migration migration + private final String fileName + private final Set excludedImageEmbedIds + + private final List textBuilders = [] + // Same-styled text runs and adjacent IF fields are buffered until a run that does not belong to the group arrives; + // a run is either text or field, so at most one of the two buffers is non-empty at any time. + private final StringBuilder text = new StringBuilder() + private String textStyleId + // The field state may be shared by consecutive paragraphs (cell, body flow) so that fields spanning paragraphs resolve. + private final FieldParseState fieldState + // POI exposes the cached result runs inside w:fldSimple as ordinary paragraph runs. Remember fields already + // resolved so that the cached «name» text does not become literal content (or get emitted once per result run). + private final Set resolvedSimpleFields = Collections.newSetFromMap(new IdentityHashMap()) + private boolean containsFieldCode = false + + ParagraphContentCollector(Migration migration, String fileName, Set excludedImageEmbedIds = Collections.emptySet(), + FieldParseState fieldState = null) { + this.migration = migration + this.fileName = fileName + this.excludedImageEmbedIds = excludedImageEmbedIds + this.fieldState = fieldState ?: new FieldParseState() + } + + void addRun(XWPFRun run, String styleId) { + if (!run.embeddedPictures.isEmpty()) { + flushText() + flushFields() + DocxImages.processRunImages(migration, run, fileName, textBuilders, excludedImageEmbedIds) + } + + CTSimpleField simpleField = run instanceof XWPFFieldRun ? (run as XWPFFieldRun).CTField : null + if (simpleField != null && !resolvedSimpleFields.contains(simpleField)) { + String instruction = simpleField.instr + if (DocxMergeFields.isMergeField(instruction)) { + resolvedSimpleFields.add(simpleField) + flushText() + flushFields() + DocxMergeFields.addMergeField(migration, textBuilders, fileName, instruction, styleId) + return + } + } + + List fieldChildren = DocxMergeFields.fieldChildren(run) + if (fieldState.depth == 0 && fieldChildren.every { DocxMergeFields.isInstructionText(it) }) { + flushFields() + // Word marks every run inside a field instruction as instrText, including the plain text of a table + // wrapped in a body-level IF field. Outside any open field such text is ordinary content. + appendText(fieldChildren.isEmpty() ? run.text() : DocxMergeFields.instructionText(fieldChildren), styleId) + return + } + containsFieldCode = true + if (fieldState.depth == 0 && fieldChildren.any { DocxMergeFields.isFieldBegin(it) }) { + flushText() + } + DocxMergeFields.handleFieldChildren(migration, fieldChildren, styleId, fieldState, textBuilders, fileName) + } + + List finish() { + flushText() + flushFields() + return textBuilders + } + + // True for a paragraph that carried only field code (instructions, cached results) and produced no content. + boolean isFieldCodeOnly() { + return containsFieldCode && textBuilders.isEmpty() + } + + private void appendText(String runText, String styleId) { + if (!runText) { + return + } + if (text.length() > 0 && styleId != textStyleId) { + flushText() + } + textStyleId = styleId + text.append(runText) + } + + private void flushText() { + if (text.length() == 0) { + return + } + DocxVariablePatterns.addText(migration, textBuilders, text.toString(), textStyleId, fileName) + text.setLength(0) + } + + private void flushFields() { + DocxMergeFields.flushPendingIfFields(migration, fieldState, textBuilders, fileName) + } +} + +static Paragraph parseParagraph(Migration migration, XWPFParagraph paragraph, String fileName, String context = null, + Set excludedImageEmbedIds = Collections.emptySet()) { + FieldParseState fieldState = new FieldParseState() + Paragraph parsed = parseFlowParagraph(migration, paragraph, fileName, fieldState, context, excludedImageEmbedIds) + warnUnterminatedField(fieldState, "paragraph: '${paragraph.text}'") + return parsed ?: buildParagraph(migration, paragraph, fileName, context, []) +} + +// Parses a paragraph that belongs to a flow of paragraphs sharing one field state (table cell, page body). Returns null +// for paragraphs that carried nothing but field code, e.g. the instruction part of an IF continued in the next paragraph. +static Paragraph parseFlowParagraph(Migration migration, XWPFParagraph paragraph, String fileName, FieldParseState fieldState, + String context = null, Set excludedImageEmbedIds = Collections.emptySet()) { + String paragraphStyleId = paragraph.styleID ?: "unknown" + ParagraphContentCollector collector = new ParagraphContentCollector(migration, fileName, excludedImageEmbedIds, fieldState) + + paragraph.runs.each { XWPFRun run -> + collector.addRun(run, captureTextStyle(migration, run, fileName, paragraphStyleId, context)) + } + + List content = collector.finish() + if (collector.fieldCodeOnly) { + return null + } + return buildParagraph(migration, paragraph, fileName, context, content) +} + +static void warnUnterminatedField(FieldParseState fieldState, String location) { + if (fieldState.depth > 0) { + println " Warning: Unterminated complex field (missing fldChar end) in ${location}" + fieldState.stack.clear() + } +} + +private static Paragraph buildParagraph(Migration migration, XWPFParagraph paragraph, String fileName, String context, + List content) { + ParagraphBuilder paragraphBuilder = new ParagraphBuilder().content(content) + String styleId = captureParagraphStyle(migration, paragraph, fileName, paragraph.styleID ?: "unknown", context) + if (styleId) { + paragraphBuilder.styleRef(styleId) + } + return paragraphBuilder.build() +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxStatistics.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxStatistics.groovy new file mode 100644 index 00000000..7a8e837f --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxStatistics.groovy @@ -0,0 +1,40 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.dto.migrationmodel.builder.DocumentObjectBuilder +import org.apache.poi.xwpf.usermodel.XWPFDocument + +static void captureDocumentStatistics(XWPFDocument doc, DocumentObjectBuilder builder) { + int shapeCount = doc.paragraphs.sum(0) { paragraph -> paragraph.runs.count { it.CTR.xmlText().contains(" table.rows.sum(0) { it.tableCells.size() } } as int + long textLength = (doc.paragraphs.sum(0) { it.text.length() } as long) + (doc.tables.sum(0) { it.text.length() } as long) + + addCountField(builder, "shapes", shapeCount) + addCountField(builder, "headers", doc.headerList.size()) + addCountField(builder, "footers", doc.footerList.size()) + addCountField(builder, "images", doc.allPictures.size()) + addCountField(builder, "paragraphs", doc.paragraphs.count { it.text.length() > 0 } as int) + addCountField(builder, "tables", doc.tables.size()) + if (doc.tables) { + builder.addCustomField("cells", cellCount.toString()) + } + addCountField(builder, "textLength", textLength) +} + +private static void addCountField(DocumentObjectBuilder builder, String name, Number count) { + if (count > 0) { + builder.addCustomField(name, count.toString()) + } +} + +static void captureDocxFileProperties(XWPFDocument doc, DocumentObjectBuilder builder) { + def coreProps = doc.properties.coreProperties + builder.addCustomField("Title", coreProps.title.toString(), coreProps.title != null) + + def extProps = doc.properties.extendedProperties + if (extProps.application) { + builder.addCustomField("Application", extProps.application) + } + if (extProps.appVersion) { + builder.addCustomField("AppVersion", extProps.appVersion) + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy new file mode 100644 index 00000000..f5509485 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy @@ -0,0 +1,103 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.Table +import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder +import com.quadient.migration.shared.TableAlignment +import com.quadient.migration.shared.TablePdfTaggingRule +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.poi.xwpf.usermodel.XWPFTableCell +import org.apache.poi.xwpf.usermodel.XWPFTableRow + +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseFlowParagraph +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseParagraph +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.warnUnterminatedField +import static com.quadient.migration.example.docx.style.DocxTableStyling.applyCellStyling +import static com.quadient.migration.example.docx.style.DocxTableStyling.applyColumnWidths +import static com.quadient.migration.example.docx.style.DocxTableStyling.applyRowHeight +import static com.quadient.migration.example.docx.util.DocxUtils.isHorizontallyMergedCell + +@Field +static final long GRID_TOLERANCE_TWIPS = 30 + +static Table parseTable(Migration migration, XWPFTable table, String fileName) { + return buildTable(migration, table, table.rows, fileName, resolveExpectedColumnCount(table), true) +} + +static int resolveExpectedColumnCount(XWPFTable table) { + def gridCols = table.CTTbl.tblGrid?.gridColList + int gridColumnCount = gridCols ? gridCols.size() : table.rows[0].tableCells.size() + int maxCellsInRow = table.rows.collect { it.tableCells.size() }.max() + return Math.max(gridColumnCount, maxCellsInRow) +} + +static Table buildTable(Migration migration, XWPFTable sourceTable, List rows, String fileName, + int expectedColumnCount, boolean useHeader) { + TableBuilder tableBuilder = createTableBuilder(migration, sourceTable, useHeader) + addRows(migration, tableBuilder, sourceTable, rows, fileName, expectedColumnCount, useHeader) + return tableBuilder.build() +} + +static TableBuilder createTableBuilder(Migration migration, XWPFTable sourceTable, boolean useHeader) { + TableBuilder tableBuilder = new TableBuilder() + .pdfTaggingRule(useHeader ? TablePdfTaggingRule.Table : TablePdfTaggingRule.None) + .alignment(TableAlignment.Center) + applyColumnWidths(tableBuilder, sourceTable) + if (migration.projectConfig.context.defaultTableStyleName) { + tableBuilder.tableStyleName(migration.projectConfig.context.defaultTableStyleName.toString()) + } + return tableBuilder +} + +// Two Word tables can be joined into one model table when their column grids match. Word lets grid widths drift by a +// few twips between otherwise identical tables, so a small tolerance is applied. +static boolean haveSameGrid(XWPFTable first, XWPFTable second) { + List firstWidths = first.CTTbl.tblGrid?.gridColList?.collect { it.w as Long } ?: [] + List secondWidths = second.CTTbl.tblGrid?.gridColList?.collect { it.w as Long } ?: [] + return firstWidths.size() == secondWidths.size() + && [firstWidths, secondWidths].transpose().every { Math.abs((it[0] as long) - (it[1] as long)) <= GRID_TOLERANCE_TWIPS } +} + +static void addRows(Migration migration, TableBuilder tableBuilder, XWPFTable sourceTable, List rows, String fileName, + int expectedColumnCount, boolean useHeader, String displayRuleId = null) { + rows.eachWithIndex { XWPFTableRow row, int ri -> + if (row.tableCells.size() != expectedColumnCount) { + println " Warning: Row ${ri + 1} has ${row.tableCells.size()} cells, expected ${expectedColumnCount}." + } + boolean isHeader = useHeader && ri == 0 && rows.size() > 1 + TableBuilder.Row rowBuilder = isHeader ? tableBuilder.addFirstHeaderRow() : tableBuilder.addRow() + if (displayRuleId) { + rowBuilder.displayRuleRef(displayRuleId) + } + String context = isHeader ? "tableHeader" : "tableRow" + + row.tableCells.eachWithIndex { XWPFTableCell cell, int ci -> + TableBuilder.Cell cellBuilder = rowBuilder.addCell() + applyCellStyling(cellBuilder, cell, sourceTable, ri, ci, rows.size(), row.tableCells.size()) + applyRowHeight(cellBuilder, row) + if (isHorizontallyMergedCell(cell)) { + cellBuilder.mergeLeft = true + } else { + cellBuilder.content(parseCellParagraphs(migration, cell, fileName, context)) + } + } + int missingCells = expectedColumnCount - rowBuilder.cells.size() + if (missingCells > 0) { + println " Adding ${missingCells} empty cells to row ${ri + 1} to match expected column count." + missingCells.times { rowBuilder.addCell().mergeLeft = true } + } + } +} + +// Paragraphs of one cell share the field state so an IF whose instruction and result sit in different paragraphs resolves. +private static List parseCellParagraphs(Migration migration, XWPFTableCell cell, String fileName, String context) { + FieldParseState fieldState = new FieldParseState() + List paragraphs = cell.paragraphs.findResults { parseFlowParagraph(migration, it, fileName, fieldState, context) } + warnUnterminatedField(fieldState, "table cell: '${cell.text}'") + if (paragraphs.isEmpty() && cell.paragraphs) { + paragraphs.add(parseParagraph(migration, cell.paragraphs.first(), fileName, context)) + } + return paragraphs +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy new file mode 100644 index 00000000..2e9b1cf0 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy @@ -0,0 +1,178 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.DocumentObject +import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef +import com.quadient.migration.api.dto.migrationmodel.PageOptions +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.builder.DocumentObjectBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.PdfMetadataBuilder +import com.quadient.migration.api.dto.migrationmodel.Table +import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder +import com.quadient.migration.shared.DocumentObjectType +import groovy.io.FileType +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFTable + +import java.util.regex.Pattern + +import static com.quadient.migration.example.docx.parser.DocxBlocks.blockName +import static com.quadient.migration.example.docx.parser.DocxBlocks.upsertBlock +import static com.quadient.migration.example.docx.parser.DocxHeadings.headingLevel +import static com.quadient.migration.example.docx.parser.DocxMergeFields.handleTableInsideField +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseFlowParagraph +import static com.quadient.migration.example.docx.parser.DocxParagraphParser.warnUnterminatedField +import static com.quadient.migration.example.docx.parser.DocxStatistics.captureDocumentStatistics +import static com.quadient.migration.example.docx.parser.DocxStatistics.captureDocxFileProperties +import static com.quadient.migration.example.docx.parser.DocxTableParser.addRows +import static com.quadient.migration.example.docx.parser.DocxTableParser.createTableBuilder +import static com.quadient.migration.example.docx.parser.DocxTableParser.parseTable +import static com.quadient.migration.example.docx.parser.DocxTableParser.resolveExpectedColumnCount + +static void parseDocxFiles(Migration migration) { + List inputFiles = [] + new File(migration.projectConfig.inputDataPath).eachFileRecurse(FileType.FILES) { File file -> + if (file.name.endsWith('.docx') && !file.name.startsWith('~$')) { + inputFiles.add(file) + } + } + inputFiles.each { parseDocxFile(migration, it) } +} + +static void parseDocxFile(Migration migration, File file) { + String fileName = file.name.substring(0, file.name.lastIndexOf('.')) + String documentType = file.parentFile.name + String relativePath = new File(migration.projectConfig.inputDataPath).toPath().relativize(file.toPath()).toString() + + println("=== Processing: " + relativePath + " ===") + DocumentObjectBuilder builder = new DocumentObjectBuilder(fileName, DocumentObjectType.Template) + .name(fileName) + .originLocations([relativePath]) + .targetFolder(resolveTargetFolder(migration, relativePath)) + .addCustomField("size", file.length().toString()) + .addCustomField("documentType", documentType) + + parsePages(migration, file, builder, fileName).each { builder.documentObjectRef(it) } + + DocumentObject documentObject = builder.build() + migration.documentObjectRepository.upsert(documentObject) +} + +private static String resolveTargetFolder(Migration migration, String relativePath) { + String[] folders = relativePath.split(Pattern.quote(File.separator)) + String subfolder = folders.length > 1 ? folders[0..-2].join(File.separator) : '' + String defaultTargetFolder = migration.projectConfig.defaultTargetFolder.toString() + return [defaultTargetFolder, subfolder].findAll().join('/') +} + +static List parsePages(Migration migration, File docxFile, DocumentObjectBuilder templateBuilder, String fileName) { + XWPFDocument doc = new XWPFDocument(new FileInputStream(docxFile)) + try { + captureDocxFileProperties(doc, templateBuilder) + setPdfMetadata(templateBuilder, doc) + DocxImages.resetImageState() + + List pages = DocxPageSections.groupIntoPages(DocxPageSections.splitIntoSections(doc)) + DocxAnchoredAreas anchoredAreas = DocxAnchoredAreas.extract(migration, doc, fileName, pages) + + int otherElements = 0 + pages.each { DocxPage page -> + String pageId = page.id(fileName) + DocxBodyContent body = new DocxBodyContent(migration, fileName, pageId) + FieldParseState fieldState = new FieldParseState(conditionalTableHandler: body.&addConditionalTable) + int floatingTables = 0 + page.bodyElements().each { elem -> + if (elem instanceof XWPFParagraph) { + Paragraph paragraph = parseFlowParagraph(migration, elem, fileName, fieldState, null, anchoredAreas.consumedEmbedIds) + if (paragraph != null) { + body.add(paragraph, headingLevel(elem)) + } + } else if (elem instanceof XWPFTable) { + if (fieldState.depth > 0) { + // The table sits inside an open IF field; it is emitted (with a display rule) once the field resolves. + handleTableInsideField(fieldState, elem) + } else if (DocxFloatingTables.isFloating(elem)) { + Area area = buildFloatingTableArea(migration, elem, page, fileName, "${pageId}_floating_table${++floatingTables}") + if (area != null) { + anchoredAreas.floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + } + } else { + addBodyTable(migration, body, elem, fileName) + } + } else { + otherElements++ + } + } + warnUnterminatedField(fieldState, "page ${page.index + 1} body") + page.content = body.sectionBlocks() + } + if (otherElements > 0) { + templateBuilder.addCustomField("otherElements", otherElements.toString()) + } + + captureDocumentStatistics(doc, templateBuilder) + + return pages.collect { buildPage(migration, doc, fileName, it, anchoredAreas) } + } finally { + doc.close() + } +} + +// A floating table is positioned absolutely, so it becomes an area of its own whose content is kept in a separate block. +private static Area buildFloatingTableArea(Migration migration, XWPFTable table, DocxPage page, String fileName, String blockId) { + Table parsedTable = parseTable(migration, table, fileName) + String firstCellText = table.rows[0]?.tableCells?.collect { it.text.trim() }?.find { it } + DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstCellText, "table", blockId), [parsedTable], fileName) + return DocxFloatingTables.buildFloatingTableArea(table, [blockRef], page.contentPosition) +} + +private static void addBodyTable(Migration migration, DocxBodyContent body, XWPFTable table, String fileName) { + int columnCount = resolveExpectedColumnCount(table) + TableBuilder tableBuilder = createTableBuilder(migration, table, true) + addRows(migration, tableBuilder, table, table.rows, fileName, columnCount, true) + body.addTable(table, tableBuilder, columnCount) +} + +private static DocumentObject buildPage(Migration migration, XWPFDocument doc, String fileName, DocxPage page, DocxAnchoredAreas anchoredAreas) { + Area mainFlowArea = new AreaBuilder() + .interactiveFlowName("Letter Content") + .content(page.content) + .position(page.contentPosition) + .flowToNextPage(true) + .build() + + int pageNumber = page.index + 1 + DocumentObjectBuilder pageBuilder = new DocumentObjectBuilder(page.id(fileName), DocumentObjectType.Page) + .internal(true) + .name("${fileName} Page ${pageNumber}") + .originLocations([fileName]) + .options(new PageOptions(page.width, page.height)) + anchoredAreas.backgroundAreasByPage[page.index]?.each { pageBuilder.area(it) } + DocxHeaderFooters.headerAreas(migration, doc, page, fileName).each { pageBuilder.area(it) } + pageBuilder.area(mainFlowArea) + DocxHeaderFooters.footerAreas(migration, doc, page, fileName).each { pageBuilder.area(it) } + anchoredAreas.floatingAreasByPage[page.index]?.each { pageBuilder.area(it) } + + DocumentObject pageObject = pageBuilder.build() + migration.documentObjectRepository.upsert(pageObject) + return pageObject +} + +// The PDF metadata of the output is taken from the document properties of the DOCX file. +static void setPdfMetadata(DocumentObjectBuilder builder, XWPFDocument doc) { + def coreProps = doc.properties.coreProperties + String title = coreProps.title?.trim() + String author = coreProps.creator?.trim() + String subject = coreProps.subject?.trim() + if (!title && !author && !subject) { + return + } + PdfMetadataBuilder pdfMetadata = new PdfMetadataBuilder() + if (title) pdfMetadata.title(title) + if (author) pdfMetadata.author(author) + if (subject) pdfMetadata.subject(subject) + builder.setPdfMetadata(pdfMetadata.build()) +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy new file mode 100644 index 00000000..b64f5217 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy @@ -0,0 +1,90 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.VariableBuilder +import com.quadient.migration.shared.DataType + +import java.util.regex.Pattern + +/** + * Converts project-configured, text-based variable markers (for example ${Name} or <>) to variable references. + * Each regular expression must contain one capturing group: group 1 becomes the variable name. + */ +class DocxVariablePatterns { + static final String CONTEXT_KEY = "docxVariablePatterns" + + static void addText(Migration migration, List textBuilders, String text, String styleId, + String fileName) { + List patterns = configuredPatterns(migration) + if (!text || patterns.isEmpty()) { + addLiteral(textBuilders, text, styleId) + return + } + + List matches = patterns.collectMany { pattern -> + def matcher = pattern.matcher(text) + def result = [] + while (matcher.find()) { + String variableId = matcher.groupCount() >= 1 ? matcher.group(1)?.trim() : null + if (variableId) { + result.add(new VariableMatch(start: matcher.start(), end: matcher.end(), variableId: variableId)) + } + } + result + }.sort { left, right -> left.start <=> right.start ?: right.end <=> left.end } + + int offset = 0 + matches.each { match -> + // First configured pattern wins for overlapping matches, avoiding duplicate variable references. + if (match.start < offset) { + return + } + addLiteral(textBuilders, text.substring(offset, match.start), styleId) + ensureVariable(migration, match.variableId, fileName) + textBuilders.add(new ParagraphBuilder.TextBuilder().variableRef(match.variableId).styleRef(styleId)) + offset = match.end + } + addLiteral(textBuilders, text.substring(offset), styleId) + } + + private static List configuredPatterns(Migration migration) { + def configured = migration.projectConfig.context?.get(CONTEXT_KEY) + Collection values = configured instanceof Collection ? configured : configured == null ? [] : [configured] + return values.collect { value -> + try { + Pattern pattern = value instanceof Pattern ? value : Pattern.compile(value.toString()) + if (pattern.matcher("").groupCount() < 1) { + println " Warning: DOCX variable pattern '${value}' has no capture group; it is ignored." + return null + } + pattern + } catch (Exception e) { + println " Warning: Invalid DOCX variable pattern '${value}': ${e.message}" + null + } + }.findAll() + } + + private static void addLiteral(List textBuilders, String text, String styleId) { + if (text) { + textBuilders.add(new ParagraphBuilder.TextBuilder().string(text).styleRef(styleId)) + } + } + + private static void ensureVariable(Migration migration, String variableId, String fileName) { + if (migration.variableRepository.find(variableId) == null) { + migration.variableRepository.upsert(new VariableBuilder(variableId) + .name(variableId) + .originLocations([fileName]) + .dataType(DataType.String) + .build()) + } + } + + private static class VariableMatch { + int start + int end + String variableId + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/report/DocxComplexityReport.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/report/DocxComplexityReport.groovy new file mode 100644 index 00000000..901c77de --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/report/DocxComplexityReport.groovy @@ -0,0 +1,48 @@ +//! --- +//! displayName: DOCX Complexity Report +//! category: Report +//! description: Creates a CSV report with statistics captured from the parsed DOCX templates, including text length and counts of paragraphs, tables, cells, images, shapes, headers and footers for each template. +//! sourceFormat: DOCX +//! --- +package com.quadient.migration.example.docx.report + +import com.quadient.migration.api.Migration +import com.quadient.migration.example.common.util.Csv +import java.nio.file.Path +import java.nio.file.Paths + +import static com.quadient.migration.example.common.util.InitMigration.initMigration + +def migration = initMigration(this.binding) +Path dstFile = Paths.get("report", "${migration.projectConfig.name}-docx-complexity-report.csv") + +run(migration, dstFile) + +static void run(Migration migration, Path documentObjectsDstPath) { + def objects = migration.documentObjectRepository.listAll().findAll { !it.internal } + + documentObjectsDstPath.toFile().createParentDirectories() + + documentObjectsDstPath.toFile().withWriter { writer -> + writer.writeLine("id,folder,name,title,textLength,paragraphs,tables,cells,images,shapes,headers,footers,byteSize") + objects.each { obj -> + + def builder = new StringBuilder() + builder.append(Csv.serialize(obj.id)) + builder.append("," + Csv.serialize(obj.targetFolder)) + builder.append("," + Csv.serialize(obj.name)) + builder.append("," + Csv.serialize((obj.customFields.get("Title") ?: ""))) + builder.append("," + (obj.customFields.get("textLength") ?: "0")) + builder.append("," + (obj.customFields.get("paragraphs") ?: "0")) + builder.append("," + (obj.customFields.get("tables") ?: "0")) + builder.append("," + (obj.customFields.get("cells") ?: "0")) + builder.append("," + (obj.customFields.get("images") ?: "0")) + builder.append("," + (obj.customFields.get("shapes") ?: "0")) + builder.append("," + (obj.customFields.get("headers") ?: "0")) + builder.append("," + (obj.customFields.get("footers") ?: "0")) + builder.append("," + (obj.customFields.get("size") ?: "0")) + writer.writeLine(builder.toString()) + } + } +} + diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxParagraphStyles.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxParagraphStyles.groovy new file mode 100644 index 00000000..80be3f06 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxParagraphStyles.groovy @@ -0,0 +1,96 @@ +package com.quadient.migration.example.docx.style + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphStyleBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphStyleDefinitionBuilder +import com.quadient.migration.shared.Alignment +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFParagraph + +import static com.quadient.migration.example.docx.style.StyleChainResolver.resolveFirst +import static com.quadient.migration.example.docx.style.StyleChainResolver.resolveParagraphPropertyChain +import static com.quadient.migration.example.docx.util.DocxUtils.calculateSHA +import static com.quadient.migration.example.docx.util.DocxUtils.twipsToSize + +static String captureParagraphStyle(Migration migration, XWPFParagraph paragraph, String fileName, String styleId, String context) { + List pPrChain = resolveParagraphPropertyChain(paragraph) + Map lineSpacing = resolveFirst(pPrChain, StyleChainResolver.&resolveLineSpacing) + + // Attribute order and fallbacks feed the style id hash; changing them invalidates persisted mappings. + LinkedHashMap styleAttributes = [ + name : styleId, + alignment : resolveFirst(pPrChain, StyleChainResolver.&resolveAlignment) ?: "LEFT", + indentationLeft : resolveFirst(pPrChain, StyleChainResolver.&resolveIndentationLeft) ?: -1, + indentationRight : resolveFirst(pPrChain, StyleChainResolver.&resolveIndentationRight) ?: -1, + indentationFirstLine: resolveFirst(pPrChain, StyleChainResolver.&resolveIndentationFirstLine) ?: -1, + spacingBefore : resolveFirst(pPrChain, StyleChainResolver.&resolveSpacingBefore) ?: -1, + spacingAfter : resolveFirst(pPrChain, StyleChainResolver.&resolveSpacingAfter) ?: -1, + lineSpacingLine : lineSpacing?.line ?: -1, + lineSpacingRule : lineSpacing?.lineRule ?: "none", + ] + + if (paragraph.numID != null) { + String numFmt = paragraph.numFmt + ?: paragraph.document.numbering?.getNum(paragraph.numID)?.CTNum?.abstractNumId?.val + styleAttributes.listId = paragraph.numID.toString() + styleAttributes.listType = numFmt + styleAttributes.listLevel = paragraph.numIlvl.toString() + styleAttributes.name = "list${numFmt ? numFmt.capitalize() : ''}_${styleAttributes.name}" + } + + String styleName = styleAttributes.name.toString() + if (context) { + styleAttributes.context = context + styleName = context + '_' + styleName + } + + String sha = calculateSHA(styleAttributes) + if (migration.paragraphStyleRepository.find(sha) == null) { + ParagraphStyleDefinitionBuilder definition = new ParagraphStyleDefinitionBuilder() + .alignment(getAlignment(styleAttributes.alignment.toString())) + .leftIndent(twipsToSize(styleAttributes.indentationLeft as long)) + .rightIndent(twipsToSize(styleAttributes.indentationRight as long)) + .firstLineIndent(twipsToSize(styleAttributes.indentationFirstLine as long)) + .spaceBefore(twipsToSize(styleAttributes.spacingBefore as long)) + .spaceAfter(twipsToSize(styleAttributes.spacingAfter as long)) + applyLineSpacing(definition, styleAttributes.lineSpacingRule.toString(), styleAttributes.lineSpacingLine as long) + + ParagraphStyleBuilder paragraphStyleBuilder = new ParagraphStyleBuilder(sha) + .name(styleName + '_' + sha.substring(0, 3) + sha.takeRight(3)) + .originLocations([fileName]) + .definition(definition.build()) + styleAttributes.each { paragraphStyleBuilder.addCustomField(it.key.toString(), it.value.toString()) } + migration.paragraphStyleRepository.upsert(paragraphStyleBuilder.build()) + } + return sha +} + +static void applyLineSpacing(ParagraphStyleDefinitionBuilder builder, String lineRule, long line) { + switch (lineRule) { + case "auto": + // Word stores "auto" line spacing in 240ths of a line. + builder.multipleOfLineSpacing(line / 240.0) + break + case "atLeast": + builder.atLeastLineSpacing(twipsToSize(line)) + break + case "exact": + builder.exactLineSpacing(twipsToSize(line)) + break + default: + builder.additionalLineSpacing(twipsToSize(-1)) + } +} + +@Field +static final Map ALIGNMENTS = [ + left : Alignment.Left, + right : Alignment.Right, + center : Alignment.Center, + justify: Alignment.JustifyLeft, + both : Alignment.JustifyLeft, +] + +static Alignment getAlignment(String alignmentAttribute) { + return ALIGNMENTS.getOrDefault(alignmentAttribute.toLowerCase(), Alignment.Left) +} \ No newline at end of file diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTableStyling.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTableStyling.groovy new file mode 100644 index 00000000..b4090725 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTableStyling.groovy @@ -0,0 +1,167 @@ +package com.quadient.migration.example.docx.style + +import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.BorderOptionsBuilder +import com.quadient.migration.shared.CellAlignment +import com.quadient.migration.shared.Color +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.poi.xwpf.usermodel.XWPFTableCell +import org.apache.poi.xwpf.usermodel.XWPFTableRow +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTBorder +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTShd +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTblBorders + +import static com.quadient.migration.example.docx.util.DocxUtils.twipsToPoints + +static void applyColumnWidths(TableBuilder tableBuilder, XWPFTable table) { + def gridCols = table.getCTTbl()?.getTblGrid()?.getGridColList() + if (!gridCols) { + return + } + List widthsTwips = gridCols.collect { it.getW() } + long totalTwips = widthsTwips.sum() as long + if (totalTwips <= 0) { + return + } + List percentages = widthsTwips.collect { it / (double) totalTwips } + // Correct any floating point rounding drift so the fractions still sum to exactly 1.0 + percentages[-1] += 1.0 - percentages.sum() + + List columnWidths = percentages.collect { new TableBuilder.ColumnWidth(Size.ofMeters(0.001), it) } + tableBuilder.columnWidths(columnWidths) +} + +static void applyRowHeight(TableBuilder.Cell cellBuilder, XWPFTableRow row) { + def trHeight = row.ctRow?.trPr?.trHeightList?.find { it.isSetVal() } + if (trHeight == null || (trHeight.val as long) <= 0) { + return + } + Size height = Size.ofPoints(twipsToPoints(trHeight.val)) + switch (trHeight.isSetHRule() ? trHeight.HRule.toString() : "auto") { + case "exact": + cellBuilder.heightFixed(height) + break + case "atLeast": + cellBuilder.heightCustom(height, Size.ofPoints(height.toPoints() * 10)) + break + default: + break // auto (or unset): let the cell grow with its content + } +} + +static void applyCellStyling(TableBuilder.Cell cellBuilder, XWPFTableCell cell, XWPFTable table, + int rowIndex = 0, int columnIndex = 0, int rowCount = 1, int columnCount = 1) { + def tblPr = table.getCTTbl()?.getTblPr() + def tcPr = cell.getCTTc()?.getTcPr() + + def tblBorders = tblPr?.getTblBorders() + def styleBorders = tableStyleBorders(table) + def tcBorders = tcPr?.getTcBorders() + Color fill = resolveShdFill(tcPr?.getShd()) ?: resolveShdFill(tblPr?.getShd()) + def tblCellMar = tblPr?.getTblCellMar() + def tcMar = tcPr?.getTcMar() + CellAlignment alignment = resolveVerticalAlignment(tcPr) + + if (fill == null && tblBorders == null && styleBorders == null && tcBorders == null && tblCellMar == null && tcMar == null && alignment == null) { + return + } + if (alignment != null) { + cellBuilder.alignment(alignment) + } + cellBuilder.border { BorderOptionsBuilder builder -> + applyBorderLine(builder.&leftLine, firstUsable(tcBorders?.getLeft(), columnIndex == 0 ? tblBorders?.getLeft() : null, + columnIndex == 0 ? styleBorders?.getLeft() : null, columnIndex > 0 ? (tblBorders?.getInsideV() ?: styleBorders?.getInsideV()) : null)) + applyBorderLine(builder.&rightLine, firstUsable(tcBorders?.getRight(), columnIndex == columnCount - 1 ? tblBorders?.getRight() : null, + columnIndex == columnCount - 1 ? styleBorders?.getRight() : null)) + applyBorderLine(builder.&topLine, firstUsable(tcBorders?.getTop(), rowIndex == 0 ? tblBorders?.getTop() : null, + rowIndex == 0 ? styleBorders?.getTop() : null, rowIndex > 0 ? (tblBorders?.getInsideH() ?: styleBorders?.getInsideH()) : null)) + applyBorderLine(builder.&bottomLine, firstUsable(tcBorders?.getBottom(), rowIndex == rowCount - 1 ? tblBorders?.getBottom() : null, + rowIndex == rowCount - 1 ? styleBorders?.getBottom() : null)) + if (fill != null) { + builder.fill(fill) + } + applyPaddingSide(builder.&paddingLeft, tcMar?.getLeft(), tblCellMar?.getLeft()) + applyPaddingSide(builder.&paddingRight, tcMar?.getRight(), tblCellMar?.getRight()) + applyPaddingSide(builder.&paddingTop, tcMar?.getTop(), tblCellMar?.getTop()) + applyPaddingSide(builder.&paddingBottom, tcMar?.getBottom(), tblCellMar?.getBottom()) + } +} + +private static CellAlignment resolveVerticalAlignment(def tcPr) { + def vAlign = tcPr?.getVAlign()?.getVal() + if (vAlign == null) { + return null + } + switch (vAlign.toString()) { + case "center": return CellAlignment.Center + case "bottom": return CellAlignment.Bottom + case "top": return CellAlignment.Top + case "both": return CellAlignment.Top // "both" (justify) has no direct equivalent; top is the closest fallback + default: return null + } +} + +private static Color resolveShdFill(CTShd shd) { + if (shd == null) { + return null + } + return StyleChainResolver.toColor(shd.getFill()) ?: StyleChainResolver.toColor(shd.getColor()) +} + +private static CTBorder firstUsable(CTBorder... borders) { + return borders.find { isUsableBorder(it) } +} + +private static CTTblBorders tableStyleBorders(XWPFTable table) { + String styleId = table.CTTbl?.tblPr?.tblStyle?.val + def styles = table.rows.findResult { it.tableCells.findResult { cell -> cell.paragraphs[0]?.document?.styles } } + Set visited = [] + while (styleId && visited.add(styleId)) { + def style = styles?.getStyle(styleId)?.CTStyle + def borders = style?.tblPr?.tblBorders + if (borders != null) return borders + styleId = style?.basedOn?.val + } + return null +} + +private static boolean isUsableBorder(CTBorder border) { + if (border == null) { + return false + } + String style = border.getVal()?.toString() + return style && style != "none" && style != "nil" +} + +private static void applyBorderLine(Closure lineSetter, CTBorder border) { + if (!isUsableBorder(border)) { + return + } + Color color = StyleChainResolver.toColor(border.getColor()) + // w:sz is in eighths of a point + Size width = border.getSz() != null ? Size.ofPoints(border.getSz().doubleValue() / 8.0) : null + lineSetter(color, width) +} + +private static void applyPaddingSide(Closure paddingSetter, def cellSideMargin, def tableSideMargin) { + Size padding = resolveMarginSize(cellSideMargin) ?: resolveMarginSize(tableSideMargin) + if (padding != null) { + paddingSetter(padding) + } +} + +private static Size resolveMarginSize(def tblWidth) { + if (tblWidth == null || !tblWidth.isSetW()) { + return null + } + String type = tblWidth.getType()?.toString() + if (type == "nil") { + return Size.ofPoints(0) + } + if (type == "pct") { + return null // percentage-based margins aren't supported here + } + Double points = twipsToPoints(tblWidth.w) + return points != null ? Size.ofPoints(points) : null +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy new file mode 100644 index 00000000..64781a0e --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy @@ -0,0 +1,56 @@ +package com.quadient.migration.example.docx.style + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.builder.TextStyleBuilder +import com.quadient.migration.api.dto.migrationmodel.builder.TextStyleDefinitionBuilder +import com.quadient.migration.shared.Size +import org.apache.poi.xwpf.usermodel.XWPFRun + +import static com.quadient.migration.example.docx.style.StyleChainResolver.resolveFirst +import static com.quadient.migration.example.docx.style.StyleChainResolver.resolveRunPropertyChain +import static com.quadient.migration.example.docx.util.DocxUtils.calculateSHA + +static String captureTextStyle(Migration migration, XWPFRun run, String fileName, String fallbackStyleId, String context) { + String styleId = run.style ?: run.CTR.RPr?.RStyleList?.find()?.val ?: fallbackStyleId + // Effective formatting is resolved by walking the full chain: direct run properties, + // then the character/paragraph style and all of its basedOn ancestors, then docDefaults. + List rPrChain = resolveRunPropertyChain(run, fallbackStyleId) + + // Attribute order and fallbacks feed the style id hash; changing them invalidates persisted mappings. + LinkedHashMap styleAttributes = [ + name : styleId, + fontName: resolveFirst(rPrChain, StyleChainResolver.&resolveFontName), + fontSize: (resolveFirst(rPrChain, StyleChainResolver.&resolveFontSize) ?: -1) as double, + bold : resolveFirst(rPrChain, StyleChainResolver.&resolveBold) ?: false, + italic : resolveFirst(rPrChain, StyleChainResolver.&resolveItalic) ?: false, + color : resolveFirst(rPrChain, StyleChainResolver.&resolveColor), + ] + + String styleName = styleId + if (context) { + styleAttributes.context = context + styleName = context + '_' + styleName + } + + String sha = calculateSHA(styleAttributes) + if (migration.textStyleRepository.find(sha) == null) { + double fontSize = styleAttributes.fontSize as double + TextStyleDefinitionBuilder definition = new TextStyleDefinitionBuilder() + .fontFamily(styleAttributes.fontName?.toString()) + .size(Size.ofPoints(fontSize == -1 ? 10 : fontSize)) + .bold(styleAttributes.bold as boolean) + .italic(styleAttributes.italic as boolean) + String color = styleAttributes.color?.toString() + if (color && color.matches("(?i)[0-9A-F]{6}")) { + definition.foregroundColor('#' + color) + } + + TextStyleBuilder textStyleBuilder = new TextStyleBuilder(sha) + .name(styleName + '_' + sha.substring(0, 3) + sha.takeRight(3)) + .originLocations([fileName]) + .definition(definition.build()) + styleAttributes.each { textStyleBuilder.addCustomField(it.key.toString(), it.value.toString()) } + migration.textStyleRepository.upsert(textStyleBuilder.build()) + } + return sha +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy new file mode 100644 index 00000000..2a626b0a --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy @@ -0,0 +1,133 @@ +package com.quadient.migration.example.docx.style + +import com.quadient.migration.shared.Color +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.poi.xwpf.usermodel.XWPFStyles +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPPrBase +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTRPr +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTStyle + +static List resolveRunPropertyChain(XWPFRun run, String fallbackStyleId) { + CTRPr directRPr = run.CTR.isSetRPr() ? run.CTR.RPr : null + XWPFStyles docStyles = run.document.styles + String styleId = directRPr?.RStyleList ? directRPr.RStyleList[0].val : (run.style ?: fallbackStyleId) + return buildPropertyChain(directRPr, docStyles, styleId, docStyles.defaultRunStyle?.RPr) { it.RPr } +} + +static List resolveParagraphPropertyChain(XWPFParagraph paragraph) { + CTPPrBase directPPr = paragraph.CTP.isSetPPr() ? paragraph.CTP.PPr : null + XWPFStyles docStyles = paragraph.document.styles + return buildPropertyChain(directPPr, docStyles, paragraph.styleID, docStyles.defaultParagraphStyle?.PPr) { it.PPr } +} + +static T resolveFirst(List chain, Closure extractor) { + return chain.findResult { extractor(it) } +} + +private static List buildPropertyChain(T directProperties, XWPFStyles docStyles, String styleId, T docDefaults, + Closure propertyExtractor) { + List chain = [] + if (directProperties != null) { + chain << directProperties + } + Set visitedStyleIds = [] + String currentStyleId = styleId + while (currentStyleId && visitedStyleIds.add(currentStyleId)) { + CTStyle ctStyle = docStyles.getStyle(currentStyleId)?.CTStyle + T properties = ctStyle != null ? propertyExtractor(ctStyle) : null + if (properties != null) { + chain << properties + } + currentStyleId = ctStyle?.isSetBasedOn() ? ctStyle.basedOn.val : null + } + if (docDefaults != null) { + chain << docDefaults + } + return chain +} + +static String resolveFontName(CTRPr rPr) { + if (rPr?.getRFontsList()) { + def font = rPr.getRFontsList().get(0) + return font.getAscii() ?: font.getHAnsi() ?: font.getEastAsia() + } + return null +} + +static Double resolveFontSize(CTRPr rPr) { + if (rPr?.getSzList()) { + def val = rPr.getSzList().get(0).getVal() + // XML 'sz' is in half-points + return val != null ? val.longValue() / 2.0 : null + } + return null +} + +static Boolean resolveBold(CTRPr rPr) { + return resolveOnOffFlag(rPr?.getBList()) +} + +static Boolean resolveItalic(CTRPr rPr) { + return resolveOnOffFlag(rPr?.getIList()) +} + +static String resolveColor(CTRPr rPr) { + if (rPr?.getColorList()) { + return colorValueToHex(rPr.getColorList().get(0).getVal()) + } + return null +} + +static String colorValueToHex(Object colorValue) { + if (colorValue instanceof byte[]) { + return colorValue.collect { String.format("%02X", it & 0xFF) }.join() + } + return colorValue?.toString() +} + +static Color toColor(Object colorValue) { + String hex = colorValueToHex(colorValue) + if (!hex || !hex.matches("(?i)[0-9A-F]{6}")) { + return null + } + return Color.fromHex('#' + hex) +} + +static String resolveAlignment(CTPPrBase pPr) { + return pPr?.getJc()?.getVal()?.toString() +} + +static Integer resolveIndentationLeft(CTPPrBase pPr) { + return (pPr?.getInd()?.isSetLeft()) ? pPr.getInd().getLeft() as int : null +} + +static Integer resolveIndentationRight(CTPPrBase pPr) { + return (pPr?.getInd()?.isSetRight()) ? pPr.getInd().getRight() as int : null +} + +static Integer resolveIndentationFirstLine(CTPPrBase pPr) { + return (pPr?.getInd()?.isSetFirstLine()) ? pPr.getInd().getFirstLine() as int : null +} + +static Integer resolveSpacingBefore(CTPPrBase pPr) { + return (pPr?.getSpacing()?.isSetBefore()) ? pPr.getSpacing().getBefore() as int : null +} + +static Integer resolveSpacingAfter(CTPPrBase pPr) { + return (pPr?.getSpacing()?.isSetAfter()) ? pPr.getSpacing().getAfter() as int : null +} + +static Map resolveLineSpacing(CTPPrBase pPr) { + if (!pPr?.getSpacing()?.isSetLine()) { + return null + } + String lineRule = pPr.getSpacing().isSetLineRule() ? pPr.getSpacing().getLineRule().toString() : "auto" + return [line: pPr.getSpacing().getLine() as long, lineRule: lineRule] +} + +private static Boolean resolveOnOffFlag(List onOffList) { + if (!onOffList) return null + def val = onOffList.get(0).getVal() + return val == null || !(val.toString() in ["false", "0", "off"]) +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/util/DocxUtils.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/util/DocxUtils.groovy new file mode 100644 index 00000000..e6a7869e --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/util/DocxUtils.groovy @@ -0,0 +1,64 @@ +package com.quadient.migration.example.docx.util + +import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.StringValue +import com.quadient.migration.shared.Size +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFTableCell + +import java.nio.charset.StandardCharsets +import java.security.MessageDigest + +@Field +static final double EMU_PER_POINT = 12700.0 + +static Double twipsToPoints(Object twips) { + if (twips == null) { + return null + } + try { + return twips.toString().toDouble() / 20.0 + } catch (NumberFormatException ignored) { + return null + } +} + +static Size twipsToSize(long twips) { + return Size.ofPoints(twips == -1 ? 0 : twips / 20.0) +} + +static Size emuToSize(long emu) { + return Size.ofPoints(emu / EMU_PER_POINT) +} + +static String sha256Hex(String value) { + return MessageDigest.getInstance("SHA-256") + .digest(value.getBytes(StandardCharsets.UTF_8)) + .encodeHex() + .toString() +} + +static String calculateSHA(LinkedHashMap attributes) { + return sha256Hex(attributes.values().collect { it.toString() }.join()).toUpperCase() +} + +static boolean isHorizontallyMergedCell(XWPFTableCell cell) { + return cell.CTTc.tcPr?.HMerge?.val?.toString() == "continue" +} + +static String extractParagraphText(DocumentContent element) { + if (!(element instanceof Paragraph)) { + return null + } + return textValues(element) + .findAll { it instanceof StringValue } + .collect { (it as StringValue).value } + .join() +} + +private static List textValues(Paragraph paragraph) { + return (paragraph.content ?: []) + .findAll { it instanceof Paragraph.Text } + .collectMany { (it as Paragraph.Text).content ?: [] } +} diff --git a/migration-examples/src/main/resources/docx-project-config-template.toml b/migration-examples/src/main/resources/docx-project-config-template.toml new file mode 100644 index 00000000..06bdd628 --- /dev/null +++ b/migration-examples/src/main/resources/docx-project-config-template.toml @@ -0,0 +1,28 @@ +name = "" +baseTemplatePath = "icm://Interactive/StandardPackage/BaseTemplates/LetterheadBaseTemplate.wfd" +#styleDefinitionPath = "icm://Interactive/StandardPackage/CompanyStyles/Styles.wfd" +inputDataPath = "" # e.g. "/migration-examples/src/main/resources/exampleResources/docx" +interactiveTenant = "StandardPackage" +#defaultTargetFolder = "" +inspireOutput = "Designer" # "Evolve", "Interactive", "Designer" +#sourceBaseTemplatePath = "icm://" +#defaultVariableStructure = "" +#defaultLanguage = "en_us" +#selectedDocumentObjectsFile = "" # e.g. "/migration-examples/document-objects/-document-objects" +#selectedDocumentObjects = [] +#subProjectId = "" + +[paths] +#images = "" +#fonts = "" +#documents = "" +#attachments = "" + +[context] +# Name of the table style applied to every parsed table. When omitted, tables +# keep only the formatting captured from the DOCX file. +#defaultTableStyleName = "" + +# Authoring conventions that should be parsed as variables in addition to native Word merge fields. +# ${...} and <<...>> +docxVariablePatterns = ['\$\{\s*([^}]+?)\s*}', '<<\s*([^>]+?)\s*>>'] diff --git a/migration-examples/src/main/resources/exampleResources/docx/01_notification_letter.docx b/migration-examples/src/main/resources/exampleResources/docx/01_notification_letter.docx new file mode 100644 index 00000000..f290c132 Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/01_notification_letter.docx differ diff --git a/migration-examples/src/main/resources/exampleResources/docx/02_terms_and_conditions.docx b/migration-examples/src/main/resources/exampleResources/docx/02_terms_and_conditions.docx new file mode 100644 index 00000000..ba63b129 Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/02_terms_and_conditions.docx differ diff --git a/migration-examples/src/main/resources/exampleResources/docx/03_spousal_assignment.docx b/migration-examples/src/main/resources/exampleResources/docx/03_spousal_assignment.docx new file mode 100644 index 00000000..99e56a7c Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/03_spousal_assignment.docx differ diff --git a/migration-examples/src/main/resources/exampleResources/docx/04_invoice_statement.docx b/migration-examples/src/main/resources/exampleResources/docx/04_invoice_statement.docx new file mode 100644 index 00000000..fc9e165a Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/04_invoice_statement.docx differ diff --git a/migration-examples/src/main/resources/exampleResources/docx/05_regulatory_disclosure.docx b/migration-examples/src/main/resources/exampleResources/docx/05_regulatory_disclosure.docx new file mode 100644 index 00000000..11a69254 Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/05_regulatory_disclosure.docx differ diff --git a/migration-examples/src/main/resources/exampleResources/docx/06_certificate.docx b/migration-examples/src/main/resources/exampleResources/docx/06_certificate.docx new file mode 100644 index 00000000..26a2ed7a Binary files /dev/null and b/migration-examples/src/main/resources/exampleResources/docx/06_certificate.docx differ