Skip to content
Merged

Docx #40

Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/)

### Added

- New parser for DOCX format

### Changed

### Fixed
Expand Down
5 changes: 5 additions & 0 deletions gradle/libs.versions.toml
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,9 @@ koin-ktor = "4.2.2"
ktor = "3.5.2"
stately = "2.1.0"

# examples
poi-ooxml = "5.5.1"

# wfd-xml
spock = "2.4-groovy-5.0"
xmlunit = "2.13.0"
Expand Down Expand Up @@ -104,6 +107,8 @@ groovy-json = { module = "org.apache.groovy:groovy-json", version.ref = "groovy"
junit-bom = { module = "org.junit:junit-bom", version.ref = "junit-jupiter" }
junit-jupiter-api = { module = "org.junit.jupiter:junit-jupiter-api", version.ref = "junit-jupiter" }
junit-jupiter-engine = { module = "org.junit.jupiter:junit-jupiter-engine", version.ref = "junit-jupiter" }
poi-ooxml = { module = "org.apache.poi:poi-ooxml", version.ref = "poi-ooxml" }
poi-ooxml-full = { module = "org.apache.poi:poi-ooxml-full", version.ref = "poi-ooxml" }

# wfd-xml
spock-core = { module = "org.spockframework:spock-core", version.ref = "spock" }
Expand Down
2 changes: 2 additions & 0 deletions migration-examples/build.gradle.kts
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,8 @@ dependencies {
implementation(libs.groovy.json)
implementation(libs.jackson.databind)
implementation(libs.slf4j.api)
implementation(libs.poi.ooxml)
implementation(libs.poi.ooxml.full)

testImplementation(platform(libs.junit.bom))
testImplementation(libs.junit.jupiter)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
//! ---
//! displayName: Parse DOCX
//! category: Parser
//! description: Parses input DOCX files specified in the project settings, translates their contents into the migration model and stores the resulting objects in the database. Persisted mappings are not applied.
//! sourceFormat: DOCX
//! ---
package com.quadient.migration.example.docx

import static com.quadient.migration.example.common.util.InitMigration.initMigration
import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles

def migration = initMigration(this.binding)

println("\nStarting Parse step...\n")
parseDocxFiles(migration)
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
//! ---
//! displayName: Parse DOCX and Apply Mapping
//! category: Parser
//! description: Parses input DOCX files specified in the project settings, translates their contents into the migration model, stores the resulting objects in the database and applies persisted mappings.
//! sourceFormat: DOCX
//! ---
package com.quadient.migration.example.docx

import static com.quadient.migration.example.common.util.InitMigration.initMigration
import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles

def migration = initMigration(this.binding)

println("\nStarting Parse step...\n")
parseDocxFiles(migration)
println("\nApplying persisted mapping...\n")
migration.mappingRepository.applyAll()
Original file line number Diff line number Diff line change
@@ -0,0 +1,232 @@
package com.quadient.migration.example.docx.parser

import com.quadient.migration.api.Migration
import com.quadient.migration.api.dto.migrationmodel.Area
import com.quadient.migration.api.dto.migrationmodel.DocumentContent
import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef
import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder
import com.quadient.migration.shared.ImageOptions
import com.quadient.migration.shared.Position
import com.quadient.migration.shared.Size
import org.apache.poi.xwpf.usermodel.XWPFDocument
import org.apache.poi.xwpf.usermodel.XWPFParagraph
import org.apache.poi.xwpf.usermodel.XWPFPictureData
import org.apache.poi.xwpf.usermodel.XWPFRun
import org.apache.xmlbeans.XmlCursor
import org.apache.xmlbeans.XmlObject
import org.openxmlformats.schemas.drawingml.x2006.picture.CTPicture
import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTP
import org.w3c.dom.Node
import org.xml.sax.InputSource

import javax.xml.parsers.DocumentBuilder
import javax.xml.parsers.DocumentBuilderFactory
import javax.xml.transform.OutputKeys
import javax.xml.transform.Transformer
import javax.xml.transform.TransformerFactory
import javax.xml.transform.dom.DOMSource
import javax.xml.transform.stream.StreamResult

import static com.quadient.migration.example.docx.parser.DocxBlocks.blockName
import static com.quadient.migration.example.docx.parser.DocxBlocks.upsertBlock
import static com.quadient.migration.example.docx.parser.DocxImages.registerImageData
import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseParagraph
import static com.quadient.migration.example.docx.util.DocxUtils.emuToSize
import static com.quadient.migration.example.docx.util.DocxUtils.extractParagraphText

class DocxAnchoredAreas {
private static final String WP_NS = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing"
private static final String W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
private static final String WPS_SHAPE_URI = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape"
private static final String PIC_NS = "http://schemas.openxmlformats.org/drawingml/2006/picture"
private static final double BACKGROUND_COVERAGE_THRESHOLD = 0.6

final Map<Integer, List<Area>> backgroundAreasByPage = [:]
final Map<Integer, List<Area>> floatingAreasByPage = [:]
final Set<String> consumedEmbedIds = []
private final Set<String> backgroundEmbedIds = []
private int textBoxes = 0

private final Migration migration
private final XWPFDocument doc
private final String fileName
private final DocumentBuilder documentBuilder
private final TransformerFactory transformerFactory = TransformerFactory.newInstance()

private DocxAnchoredAreas(Migration migration, XWPFDocument doc, String fileName) {
this.migration = migration
this.doc = doc
this.fileName = fileName
DocumentBuilderFactory dbf = DocumentBuilderFactory.newInstance()
dbf.namespaceAware = true
this.documentBuilder = dbf.newDocumentBuilder()
}

static DocxAnchoredAreas extract(Migration migration, XWPFDocument doc, String fileName, List<DocxPage> pages) {
DocxAnchoredAreas result = new DocxAnchoredAreas(migration, doc, fileName)
// Backgrounds are resolved over the whole document first so the same picture is never emitted as floating too.
pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addBackgroundArea(it, page) } }
pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) } }
return result
}

private void eachUniqueAnchor(DocxPage page, Closure<Void> handler) {
Set<Object> seenDrawingIds = []
page.paragraphs().each { XWPFParagraph paragraph ->
paragraph.runs.each { XWPFRun run ->
findAnchors(run).each { CTAnchor anchor ->
if (seenDrawingIds.add(anchor.docPr?.id ?: anchor.xmlText())) {
handler(anchor)
}
}
}
}
}

private void addBackgroundArea(CTAnchor anchor, DocxPage page) {
if (!isPicture(anchor) || !coversPage(anchor, page)) {
return
}
String embedId = extractBlipEmbedId(anchor)
Position position = new Position(Size.ofPoints(0), Size.ofPoints(0), page.width, page.height)
Area area = embedId ? buildImageArea(anchor, embedId, position) : null
if (area != null) {
backgroundAreasByPage.computeIfAbsent(page.index) { [] }.add(area)
backgroundEmbedIds.add(embedId)
consumedEmbedIds.add(embedId)
}
}

private void addFloatingArea(CTAnchor anchor, DocxPage page) {
Area area = null
if (anchor.graphic?.graphicData?.uri == WPS_SHAPE_URI) {
area = buildTextBoxArea(anchor, page)
} else if (isPicture(anchor)) {
String embedId = extractBlipEmbedId(anchor)
if (embedId && !backgroundEmbedIds.contains(embedId)) {
area = buildImageArea(anchor, embedId, resolveAnchorPosition(anchor, page))
if (area != null) {
consumedEmbedIds.add(embedId)
}
}
}
if (area != null) {
floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area)
}
}

private static boolean isPicture(CTAnchor anchor) {
return anchor.graphic?.graphicData?.uri == PIC_NS
}

private static boolean coversPage(CTAnchor anchor, DocxPage page) {
Size width = emuToSize(anchor.extent?.cx ?: 0L)
Size height = emuToSize(anchor.extent?.cy ?: 0L)
return width.toPoints() >= page.width.toPoints() * BACKGROUND_COVERAGE_THRESHOLD
&& height.toPoints() >= page.height.toPoints() * BACKGROUND_COVERAGE_THRESHOLD
}

private static List<CTAnchor> findAnchors(XWPFRun run) {
XmlObject[] found = run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:anchor")
return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) }
}

private static String extractBlipEmbedId(CTAnchor anchor) {
XmlObject[] pics = anchor.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic")
if (pics.length == 0) {
return null
}
CTPicture pic = pics[0] instanceof CTPicture ? pics[0] as CTPicture : CTPicture.Factory.parse(pics[0].toString())
return pic.blipFill?.blip?.embed
}

private Area buildImageArea(CTAnchor anchor, String embedId, Position position) {
XWPFPictureData data = doc.getPictureDataByID(embedId)
if (data == null) {
return null
}
Size width = emuToSize(anchor.extent?.cx ?: 0L)
Size height = emuToSize(anchor.extent?.cy ?: 0L)
String imageId = registerImageData(migration, data, fileName, new ImageOptions(width, height))
if (imageId == null) {
return null
}
return new AreaBuilder().imageRef(imageId).position(position).build()
}

private Area buildTextBoxArea(CTAnchor anchor, DocxPage page) {
XmlCursor cursor = anchor.graphic.graphicData.newCursor()
try {
if (!cursor.toFirstChild()) {
return null
}
def shapeDom = documentBuilder.parse(new InputSource(new StringReader(cursor.xmlText())))
def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent")
if (txbxContentNodes.length == 0) {
return null
}
List<DocumentContent> contentItems = []
def children = txbxContentNodes.item(0).childNodes
for (int i = 0; i < children.length; i++) {
Node child = children.item(i)
if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') {
XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc)
contentItems.add(parseParagraph(migration, paragraph, fileName))
}
}
if (contentItems.isEmpty()) {
return null
}
String blockId = "${page.id(fileName)}_textbox${++textBoxes}"
String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null }
DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName)
return new AreaBuilder().content([blockRef]).position(resolveAnchorPosition(anchor, page)).build()
} finally {
cursor.dispose()
}
}

private CTP domParagraphToCtp(Node pNode) {
StringWriter sw = new StringWriter()
Transformer transformer = transformerFactory.newTransformer()
transformer.setOutputProperty(OutputKeys.OMIT_XML_DECLARATION, "yes")
transformer.transform(new DOMSource(pNode), new StreamResult(sw))
String fragmentXml = sw.toString().replaceFirst(/^<w:p /, '<xml-fragment ').replaceFirst(/<\/w:p>$/, '</xml-fragment>')
return CTP.Factory.parse(fragmentXml)
}

static Position resolveAnchorPosition(CTAnchor anchor, DocxPage page) {
Size extentWidth = emuToSize(anchor.extent?.cx ?: 0L)
Size extentHeight = emuToSize(anchor.extent?.cy ?: 0L)
double marginLeft = page.contentPosition.x.toPoints()
double marginTop = page.contentPosition.y.toPoints()
boolean relativeToPageH = anchor.positionH?.relativeFrom?.toString() == "page"
boolean relativeToPageV = anchor.positionV?.relativeFrom?.toString() == "page"

double x
if (anchor.positionH?.isSetPosOffset()) {
x = emuToSize(anchor.positionH.posOffset).toPoints() + (relativeToPageH ? 0.0d : marginLeft)
} else {
double leftBound = relativeToPageH ? 0.0d : marginLeft
double rightBound = relativeToPageH ? page.width.toPoints() : marginLeft + page.contentPosition.width.toPoints()
switch (anchor.positionH?.align?.toString()) {
case "right":
case "outside":
x = rightBound - extentWidth.toPoints()
break
case "center":
x = leftBound + (rightBound - leftBound - extentWidth.toPoints()) / 2.0d
break
default:
x = leftBound
}
}

double y = anchor.positionV?.isSetPosOffset()
? emuToSize(anchor.positionV.posOffset).toPoints() + (relativeToPageV ? 0.0d : marginTop)
: marginTop

return new Position(Size.ofPoints(x), Size.ofPoints(y), extentWidth, extentHeight)
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
package com.quadient.migration.example.docx.parser

import com.quadient.migration.api.Migration
import com.quadient.migration.api.dto.migrationmodel.DisplayRuleRef
import com.quadient.migration.api.dto.migrationmodel.DocumentContent
import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef
import com.quadient.migration.api.dto.migrationmodel.builder.DocumentObjectBuilder
import com.quadient.migration.shared.DocumentObjectType
import groovy.transform.Field

@Field
static final int MAX_BLOCK_NAME_LENGTH = 60

// Stores a piece of content as an internal block and returns the reference to place in the flow instead.
static DocumentObjectRef upsertBlock(Migration migration, String id, String name, List<DocumentContent> content, String fileName,
String displayRuleId = null) {
migration.documentObjectRepository.upsert(new DocumentObjectBuilder(id, DocumentObjectType.Block)
.name(name)
.content(content)
.internal(true)
.originLocations([fileName])
.build())
return new DocumentObjectRef(id, displayRuleId ? new DisplayRuleRef(displayRuleId) : null)
}

// Derives a readable block name from source text (a heading, a table's first cell, ...), shortened at a word boundary.
static String blockName(String text, String suffix, String fallback) {
String normalized = text?.replaceAll(/\s+/, " ")?.trim()
if (!normalized) {
return fallback
}
if (normalized.length() > MAX_BLOCK_NAME_LENGTH) {
normalized = normalized.substring(0, MAX_BLOCK_NAME_LENGTH).replaceFirst(/\s+\S*$/, "")
}
return suffix ? "${normalized} ${suffix}" : normalized
}
Loading
Loading