This commit is contained in:
jstaerk
2020-11-11 21:15:49 +01:00
parent 442e061607
commit e1881a3891
10 changed files with 193 additions and 140 deletions

View File

@@ -51,6 +51,7 @@ switch
- contacts also for recipients
- absolute and relative allowances and charges on item and document level #135,
- support contact fax numbers
- closes #190 BOM not treated correctly on XML input file
### 2.0 still todo
- dont show empty tax number field

View File

@@ -1,9 +1,25 @@
package org.mustangproject;
import java.io.ByteArrayInputStream;
import java.io.IOException;
import java.io.UnsupportedEncodingException;
import java.math.BigDecimal;
import java.math.RoundingMode;
import java.util.logging.Level;
import java.util.logging.Logger;
public class XMLTools {
import org.dom4j.io.XMLWriter;
import org.mustangproject.ZUGFeRD.ZUGFeRD2PullProvider;
public class XMLTools extends XMLWriter {
public String escapeAttributeEntities(String s) {
return super.escapeAttributeEntities(s);
}
public String escapeElementEntities(String s) {
return super.escapeElementEntities(s);
}
public static String nDigitFormat(BigDecimal value, int scale) {
@@ -41,9 +57,15 @@ public class XMLTools {
sb.append("�"); // Unicode replacement character
} else {
switch (c) {
case '&': sb.append("&"); break;
case '>': sb.append(">"); break;
case '<': sb.append("&lt;"); break;
case '&':
sb.append("&amp;");
break;
case '>':
sb.append("&gt;");
break;
case '<':
sb.append("&lt;");
break;
// Uncomment next two if encoding for an XML attribute
// case '\'' sb.append("&apos;"); break;
// case '\"' sb.append("&quot;"); break;
@@ -52,7 +74,8 @@ public class XMLTools {
// case '\r' sb.append("&#13;"); break;
// case '\t' sb.append("&#9;"); break;
default: sb.append((char)c);
default:
sb.append((char) c);
}
}
} else if ((c >= 0xd800 && c <= 0xdfff) || c == 0xfffe || c == 0xffff) {
@@ -66,4 +89,53 @@ public class XMLTools {
}
return sb.toString();
}
/**
* Returns the Byte Order Mark size and thus allows to skips over a BOM
* at the beginning of the given ByteArrayInputStream, if one exists.
*
* @param is the ByteArrayInputStream used
* @throws IOException if can not be read from is
* @see <a href="https://www.w3.org/TR/xml/#sec-guessing">Autodetection of Character Encodings</a>
*/
public static int guessBOMSize(ByteArrayInputStream is) throws IOException {
byte[] pad = new byte[4];
is.read(pad);
is.reset();
int test2 = ((pad[0] & 0xFF) << 8) | (pad[1] & 0xFF);
int test3 = ((test2 & 0xFFFF) << 8) | (pad[2] & 0xFF);
int test4 = ((test3 & 0xFFFFFF) << 8) | (pad[3] & 0xFF);
//
if (test4 == 0x0000FEFF || test4 == 0xFFFE0000 || test4 == 0x0000FFFE || test4 == 0xFEFF0000) {
// UCS-4: BOM takes 4 bytes
return 4;
} else if (test3 == 0xEFBBFF) {
// UTF-8: BOM takes 3 bytes
return 3;
} else if (test2 == 0xFEFF || test2 == 0xFFFE) {
// UTF-16: BOM takes 2 bytes
return 2;
}
return 0;
}
/***
* removes utf8 byte order marks from byte arrays, in case one is there
* @param zugferdRaw
* @return the byte array without bom
*/
public static byte[] removeBOM(byte[] zugferdRaw) {
byte[] zugferdData;
if ((zugferdRaw[0] == (byte) 0xEF) && (zugferdRaw[1] == (byte) 0xBB) && (zugferdRaw[2] == (byte) 0xBF)) {
// I don't like BOMs, lets remove it
zugferdData = new byte[zugferdRaw.length - 3];
System.arraycopy(zugferdRaw, 3, zugferdData, 0, zugferdRaw.length - 3);
} else {
zugferdData = zugferdRaw;
}
return zugferdData;
}
}

View File

@@ -1,21 +1,23 @@
/** **********************************************************************
*
/**
* *********************************************************************
* <p>
* Copyright 2018 Jochen Staerk
*
* <p>
* Use is subject to license terms.
*
* <p>
* Licensed under the Apache License, Version 2.0 (the "License"); you may not
* use this file except in compliance with the License. You may obtain a copy
* of the License at http://www.apache.org/licenses/LICENSE-2.0.
*
* <p>
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
* WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
*
* <p>
* See the License for the specific language governing permissions and
* limitations under the License.
*
*********************************************************************** */
* <p>
* **********************************************************************
*/
package org.mustangproject.ZUGFeRD;
import org.dom4j.Document;
@@ -52,6 +54,7 @@ public class ZUGFeRD1PullProvider extends ZUGFeRD2PullProvider implements IXMLPr
@Override
public void setTest() {
}
private String vatFormat(BigDecimal value) {
return XMLTools.nDigitFormat(value, 2);
}
@@ -404,14 +407,7 @@ public class ZUGFeRD1PullProvider extends ZUGFeRD2PullProvider implements IXMLPr
byte[] zugferdRaw;
try {
zugferdRaw = xml.getBytes("UTF-8");
if ((zugferdRaw[0] == (byte) 0xEF) && (zugferdRaw[1] == (byte) 0xBB) && (zugferdRaw[2] == (byte) 0xBF)) {
// I don't like BOMs, lets remove it
zugferdData = new byte[zugferdRaw.length - 3];
System.arraycopy(zugferdRaw, 3, zugferdData, 0, zugferdRaw.length - 3);
} else {
zugferdData = zugferdRaw;
}
zugferdData = XMLTools.removeBOM(zugferdRaw);
} catch (UnsupportedEncodingException e) {
Logger.getLogger(ZUGFeRD1PullProvider.class.getName()).log(Level.SEVERE, null, e);
}

View File

@@ -615,13 +615,7 @@ public class ZUGFeRD2PullProvider implements IXMLProvider {
try {
zugferdRaw = xml.getBytes("UTF-8");
if ((zugferdRaw[0] == (byte) 0xEF) && (zugferdRaw[1] == (byte) 0xBB) && (zugferdRaw[2] == (byte) 0xBF)) {
// I don't like BOMs, lets remove it
zugferdData = new byte[zugferdRaw.length - 3];
System.arraycopy(zugferdRaw, 3, zugferdData, 0, zugferdRaw.length - 3);
} else {
zugferdData = zugferdRaw;
}
zugferdData=XMLTools.removeBOM(zugferdRaw);
} catch (UnsupportedEncodingException e) {
Logger.getLogger(ZUGFeRD2PullProvider.class.getName()).log(Level.SEVERE, null, e);
}

View File

@@ -161,8 +161,8 @@
<!-- http://stackoverflow.com/questions/574594/how-can-i-create-an-executable-jar-with-dependencies-using-maven
mvn clean compile assembly:single -->
<!-- or whatever version you use -->
<source>1.7</source>
<target>1.7</target>
<source>8</source>
<target>8</target>
</configuration>
</plugin>
<!-- /ZUV -->

View File

@@ -1,5 +1,6 @@
package org.mustangproject.validator;
import org.mustangproject.XMLTools;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;

View File

@@ -1,14 +0,0 @@
package org.mustangproject.validator;
import org.dom4j.io.XMLWriter;
public class XMLTools extends XMLWriter {
public String escapeAttributeEntities(String s) {
return super.escapeAttributeEntities(s);
}
public String escapeElementEntities(String s) {
return super.escapeElementEntities(s);
}
}

View File

@@ -18,6 +18,7 @@ import javax.xml.xpath.XPathConstants;
import javax.xml.xpath.XPathExpression;
import javax.xml.xpath.XPathFactory;
import org.mustangproject.XMLTools;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import org.w3c.dom.Document;
@@ -61,7 +62,7 @@ public class XMLValidator extends Validator {
// file existence must have been checked before
try {
zfXML = new String(Files.readAllBytes(Paths.get(name)));
zfXML = new String(XMLTools.removeBOM(Files.readAllBytes(Paths.get(name))));
} catch (IOException e) {
ValidationResultItem vri = new ValidationResultItem(ESeverity.exception, e.getMessage()).setSection(9)
@@ -112,8 +113,6 @@ public class XMLValidator extends Validator {
failedRules = 0;
ByteArrayInputStream xmlByteInputStream = new ByteArrayInputStream(zfXML.getBytes(StandardCharsets.UTF_8));
if (zfXML.isEmpty()) {
ValidationResultItem res = new ValidationResultItem(ESeverity.exception,
"XML data not found in " + filename
@@ -145,8 +144,8 @@ public class XMLValidator extends Validator {
// document.getElementsByTagNameNS("*",...
DocumentBuilder db = dbf.newDocumentBuilder();
Document doc = db.parse(xmlByteInputStream);
InputSource is = new InputSource(new StringReader(zfXML));
Document doc = db.parse(is);
Element root = doc.getDocumentElement();

View File

@@ -1,11 +1,7 @@
package org.mustangproject.validator;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileNotFoundException;
import java.io.IOException;
import java.io.InputStream;
import java.io.StringWriter;
import java.io.*;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.security.MessageDigest;
@@ -22,11 +18,13 @@ import org.dom4j.DocumentException;
import org.dom4j.DocumentHelper;
import org.dom4j.io.OutputFormat;
import org.dom4j.io.XMLWriter;
import org.mustangproject.XMLTools;
import org.riversun.bigdoc.bin.BigFileSearcher;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import org.w3c.dom.Document;
import org.w3c.dom.Element;
import org.xml.sax.InputSource;
import org.xml.sax.SAXParseException;
//abstract class
@@ -166,7 +164,11 @@ public class ZUGFeRDValidator {
DocumentBuilderFactory dbf = DocumentBuilderFactory.newInstance();
DocumentBuilder db = dbf.newDocumentBuilder();
Document doc = db.parse(file);
byte[] content=Files.readAllBytes(file.toPath());
content= XMLTools.removeBOM(content);
String s=new String(content);
InputSource is = new InputSource(new StringReader(s));
Document doc = db.parse(is);
Element root = doc.getDocumentElement();
isXML=true;//no exception so far
@@ -176,6 +178,8 @@ public class ZUGFeRDValidator {
// probably no xml file, sth like SAXParseException content not allowed in prolog
// ignore isXML is already false
// in the tests, this may error-out anyway
//ex.printStackTrace();
}
if (isXML) {
pdfValidity = true;

View File

@@ -1,4 +1,4 @@
<?xml version="1.0" encoding="UTF-8"?>
<?xml version="1.0" encoding="UTF-8"?><!-- the document starts with an invisible utf8 byte order mark, which is valid-->
<rsm:CrossIndustryInvoice xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:rsm="urn:un:unece:uncefact:data:standard:CrossIndustryInvoice:100" xmlns:ram="urn:un:unece:uncefact:data:standard:ReusableAggregateBusinessInformationEntity:100" xmlns:udt="urn:un:unece:uncefact:data:standard:UnqualifiedDataType:100">
<rsm:ExchangedDocumentContext>
<ram:GuidelineSpecifiedDocumentContextParameter>