whitespace corrections+switch to ioutils

This commit is contained in:
jstaerk
2024-07-10 11:06:12 +02:00
parent acbfa2c836
commit c88ec268f5
3 changed files with 165 additions and 169 deletions

View File

@@ -5,6 +5,7 @@
- Fix #389: ClassCastException: ZUGFeRDExporterFromA3 - Fix #389: ClassCastException: ZUGFeRDExporterFromA3
- jakarta support #372 - jakarta support #372
- Upgrade to PDFBox 3 #373 - Upgrade to PDFBox 3 #373
- Requires Java 11
- #397 - #397
- #392 CLI: action combine: --ignorefileextension to ignore PDF/A input file errors dosen't work - #392 CLI: action combine: --ignorefileextension to ignore PDF/A input file errors dosen't work
- for CLI combine, fx is now default - for CLI combine, fx is now default

View File

@@ -14,10 +14,7 @@ package org.mustangproject.ZUGFeRD;
* @author jstaerk * @author jstaerk
*/ */
import java.io.BufferedInputStream; import java.io.*;
import java.io.ByteArrayInputStream;
import java.io.IOException;
import java.io.InputStream;
import java.math.BigDecimal; import java.math.BigDecimal;
import java.nio.charset.StandardCharsets; import java.nio.charset.StandardCharsets;
import java.nio.file.Files; import java.nio.file.Files;
@@ -43,8 +40,8 @@ import javax.xml.xpath.XPathExpression;
import javax.xml.xpath.XPathExpressionException; import javax.xml.xpath.XPathExpressionException;
import javax.xml.xpath.XPathFactory; import javax.xml.xpath.XPathFactory;
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.Loader; import org.apache.pdfbox.Loader;
import org.apache.pdfbox.io.IOUtils;
import org.apache.pdfbox.pdmodel.PDDocument; import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary; import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary;
import org.apache.pdfbox.pdmodel.PDEmbeddedFilesNameTreeNode; import org.apache.pdfbox.pdmodel.PDEmbeddedFilesNameTreeNode;
@@ -786,11 +783,11 @@ public class ZUGFeRDImporter {
static String convertStreamToString(java.io.InputStream is) { static String convertStreamToString(java.io.InputStream is) {
// TODO wouldn't we use IOUtils.toByteArray nowadays??? try {
// source https://stackoverflow.com/questions/309424/how-do-i-read-convert-an-inputstream-into-a-string-in-java referring to return IOUtils.toString(is, StandardCharsets.UTF_8);
// https://community.oracle.com/blogs/pat/2004/10/23/stupid-scanner-tricks } catch (IOException e) {
final Scanner s = new Scanner(is, StandardCharsets.UTF_8).useDelimiter("\\A"); throw new UncheckedIOException(e);
return s.hasNext() ? s.next() : ""; }
} }
/** /**

View File

@@ -75,132 +75,132 @@ public class ZUGFeRDValidator {
return wasCompletelyValid; return wasCompletelyValid;
} }
private String internalValidate (String contextFilename, InputStream inputStream, long inputLength) {
context.clear();
StringBuilder finalStringResult = new StringBuilder();
SimpleDateFormat isoDF = new SimpleDateFormat("yyyy-MM-dd HH:mm:ss");
Date date = new Date();
startTime = Calendar.getInstance().getTimeInMillis();
context.setFilename(contextFilename);// fallback to provided name
finalStringResult.append("<validation filename='").append (contextFilename).append ("' datetime='").append (isoDF.format(date)).append ("'>");
boolean isPDF = false; private String internalValidate(String contextFilename, InputStream inputStream, long inputLength) {
byte[] content = null; context.clear();
try { StringBuilder finalStringResult = new StringBuilder();
SimpleDateFormat isoDF = new SimpleDateFormat("yyyy-MM-dd HH:mm:ss");
Date date = new Date();
startTime = Calendar.getInstance().getTimeInMillis();
context.setFilename(contextFilename);// fallback to provided name
finalStringResult.append("<validation filename='").append(contextFilename).append("' datetime='").append(isoDF.format(date)).append("'>");
if (contextFilename == null || contextFilename.isEmpty ()) { boolean isPDF = false;
optionsRecognized = false; byte[] content = null;
context.addResultItem(new ValidationResultItem(ESeverity.fatal, "Filename not specified").setSection(10) try {
.setPart(EPart.pdf));
}
PDFValidator pdfv = new PDFValidator(context); if (contextFilename == null || contextFilename.isEmpty()) {
if (inputStream == null) { optionsRecognized = false;
context.addResultItem( context.addResultItem(new ValidationResultItem(ESeverity.fatal, "Filename not specified").setSection(10)
new ValidationResultItem(ESeverity.fatal, "File not found").setSection(1).setPart(EPart.pdf)); .setPart(EPart.pdf));
} else if (inputLength < 32) { }
// with less than 32 bytes it can not even be a proper XML file
// Except it is "<?xml version='1.0'?><xml/>" LOL
context.addResultItem(
new ValidationResultItem(ESeverity.fatal, "File too small").setSection(5).setPart(EPart.pdf));
} else if (inputLength >= Integer.MAX_VALUE) {
// Byte arrays are limited to 2GB in Java
context.addResultItem(
new ValidationResultItem(ESeverity.fatal, "File too big").setSection(5).setPart(EPart.pdf));
} else {
content = IOUtils.toByteArray(inputStream);
XMLValidator xv = new XMLValidator(context);
if (disableNotices) {
xv.disableNotices();
}
isPDF = ByteArraySearcher.startsWith(content, new byte[] {'%', 'P', 'D', 'F'});
if (isPDF) {
// Avoid reading again from file
pdfv.setFilenameAndContents(contextFilename, content);
optionsRecognized = true; PDFValidator pdfv = new PDFValidator(context);
finalStringResult.append("<pdf>"); if (inputStream == null) {
try { context.addResultItem(
pdfv.validate(); new ValidationResultItem(ESeverity.fatal, "File not found").setSection(1).setPart(EPart.pdf));
} else if (inputLength < 32) {
// with less than 32 bytes it can not even be a proper XML file
// Except it is "<?xml version='1.0'?><xml/>" LOL
context.addResultItem(
new ValidationResultItem(ESeverity.fatal, "File too small").setSection(5).setPart(EPart.pdf));
} else if (inputLength >= Integer.MAX_VALUE) {
// Byte arrays are limited to 2GB in Java
context.addResultItem(
new ValidationResultItem(ESeverity.fatal, "File too big").setSection(5).setPart(EPart.pdf));
} else {
content = IOUtils.toByteArray(inputStream);
XMLValidator xv = new XMLValidator(context);
if (disableNotices) {
xv.disableNotices();
}
isPDF = ByteArraySearcher.startsWith(content, new byte[]{'%', 'P', 'D', 'F'});
if (isPDF) {
// Avoid reading again from file
pdfv.setFilenameAndContents(contextFilename, content);
sha1Checksum = calcSHA1(content); optionsRecognized = true;
finalStringResult.append("<pdf>");
try {
pdfv.validate();
// Validate PDF sha1Checksum = calcSHA1(content);
getPdfValidationResults(finalStringResult, pdfv, xv); // Validate PDF
} catch (IrrecoverableValidationError irx) {
LOGGER.info(irx.getMessage());
}
finalStringResult.append("</pdf>\n"); getPdfValidationResults(finalStringResult, pdfv, xv);
} catch (IrrecoverableValidationError irx) {
LOGGER.info(irx.getMessage());
}
context.clearCustomXML(); finalStringResult.append("</pdf>\n");
} else {
boolean isXML = false;
String xmlAsString = null;
try {
DocumentBuilderFactory dbf = DocumentBuilderFactory.newInstance();
DocumentBuilder db = dbf.newDocumentBuilder();
content = XMLTools.removeBOM(content); context.clearCustomXML();
xmlAsString = new String(content, StandardCharsets.UTF_8); } else {
InputSource is = new InputSource(new StringReader(xmlAsString)); boolean isXML = false;
Document doc = db.parse(is); String xmlAsString = null;
try {
DocumentBuilderFactory dbf = DocumentBuilderFactory.newInstance();
DocumentBuilder db = dbf.newDocumentBuilder();
Element root = doc.getDocumentElement(); content = XMLTools.removeBOM(content);
isXML = true;//no exception so far xmlAsString = new String(content, StandardCharsets.UTF_8);
InputSource is = new InputSource(new StringReader(xmlAsString));
Document doc = db.parse(is);
} catch (Exception ex) { Element root = doc.getDocumentElement();
// probably no xml file, sth like SAXParseException content not allowed in prolog isXML = true;//no exception so far
// ignore isXML is already false
// in the tests, this may error-out anyway
LOGGER.info("No XML part provided");
}
if (isXML) {
pdfValidity = true;
optionsRecognized = true;
xv.setStringContent (xmlAsString);
xv.setAutoload(false);
xv.setFilename(contextFilename);
sha1Checksum = calcSHA1(content);
displayXMLValidationOutput = true; } catch (Exception ex) {
// probably no xml file, sth like SAXParseException content not allowed in prolog
// ignore isXML is already false
// in the tests, this may error-out anyway
LOGGER.info("No XML part provided");
}
if (isXML) {
pdfValidity = true;
optionsRecognized = true;
xv.setStringContent(xmlAsString);
xv.setAutoload(false);
xv.setFilename(contextFilename);
sha1Checksum = calcSHA1(content);
} else { displayXMLValidationOutput = true;
optionsRecognized = false;
context.addResultItem(new ValidationResultItem(ESeverity.exception,
"File does not look like PDF nor XML (contains neither %PDF nor <?xml)").setSection(8));
} } else {
} optionsRecognized = false;
if ((optionsRecognized) && (displayXMLValidationOutput)) { context.addResultItem(new ValidationResultItem(ESeverity.exception,
finalStringResult.append("<xml>"); "File does not look like PDF nor XML (contains neither %PDF nor <?xml)").setSection(8));
try {
xv.validate();
} catch (IrrecoverableValidationError irx) {
LOGGER.info("The hell");
}
finalStringResult.append(xv.getXMLResult());
finalStringResult.append("</xml>");
context.clearCustomXML();
}
if ((isPDF) && (!pdfValidity)) { }
context.setInvalid(); }
} if ((optionsRecognized) && (displayXMLValidationOutput)) {
finalStringResult.append("<xml>");
try {
xv.validate();
} catch (IrrecoverableValidationError irx) {
LOGGER.error("XML validation threw an exception ", irx);
}
finalStringResult.append(xv.getXMLResult());
finalStringResult.append("</xml>");
context.clearCustomXML();
}
} if ((isPDF) && (!pdfValidity)) {
} catch (IrrecoverableValidationError | IOException irx) { context.setInvalid();
LOGGER.info(irx.getMessage()); }
context.setInvalid ();
} finally {
finalStringResult.append(context.getXMLResult());
finalStringResult.append("</validation>");
} }
} catch (IrrecoverableValidationError | IOException irx) {
LOGGER.info(irx.getMessage());
context.setInvalid();
} finally {
finalStringResult.append(context.getXMLResult());
finalStringResult.append("</validation>");
return formatOutput(finalStringResult, isPDF); }
return formatOutput(finalStringResult, isPDF);
} }
/*** /***
@@ -210,61 +210,59 @@ public class ZUGFeRDValidator {
* @return a xml string with the validation result * @return a xml string with the validation result
*/ */
public String validate(String filename) { public String validate(String filename) {
String contextFilename; String contextFilename;
InputStream inputStream; InputStream inputStream;
long inputLength; long inputLength;
if (filename == null) { if (filename == null) {
// No filename provided // No filename provided
contextFilename = ""; contextFilename = "";
inputStream = null; inputStream = null;
inputLength = 0; inputLength = 0;
} else { } else {
File file = new File(filename); File file = new File(filename);
// set filename without path // set filename without path
contextFilename = file.getName (); contextFilename = file.getName();
if (file.isFile ()) { if (file.isFile()) {
try { try {
inputStream = new FileInputStream (file); inputStream = new FileInputStream(file);
inputLength = Files.size (file.toPath ()); inputLength = Files.size(file.toPath());
} catch (IOException ex) { } catch (IOException ex) {
throw new UncheckedIOException (ex); throw new UncheckedIOException(ex);
} }
} else { } else {
// Non-existing or Directory // Non-existing or Directory
inputStream = null; inputStream = null;
inputLength = 0; inputLength = 0;
} }
} }
try { try {
return internalValidate (contextFilename, inputStream, inputLength); return internalValidate(contextFilename, inputStream, inputLength);
} finally { } finally {
StreamHelper.close (inputStream); StreamHelper.close(inputStream);
} }
} }
public String validate(InputStream inputStream, String fileNameOfInputStream) { public String validate(InputStream inputStream, String fileNameOfInputStream) {
long inputLength; long inputLength;
try { try {
inputLength = inputStream == null ? 0 : inputStream.available (); inputLength = inputStream == null ? 0 : inputStream.available();
} } catch (IOException ex) {
catch (IOException ex) { throw new UncheckedIOException(ex);
throw new UncheckedIOException (ex); }
} try {
try { return internalValidate(fileNameOfInputStream, inputStream, inputLength);
return internalValidate (fileNameOfInputStream, inputStream, inputLength); } finally {
} finally { StreamHelper.close(inputStream);
StreamHelper.close (inputStream); }
} }
}
public String validate(byte[] bytes, String fileNameOfInputStream) { public String validate(byte[] bytes, String fileNameOfInputStream) {
try(ByteArrayInputStream bais = new ByteArrayInputStream (bytes)) { try (ByteArrayInputStream bais = new ByteArrayInputStream(bytes)) {
return internalValidate (fileNameOfInputStream, bais, bytes.length); return internalValidate(fileNameOfInputStream, bais, bytes.length);
} } catch (IOException ex) {
catch (IOException ex) { throw new UncheckedIOException(ex);
throw new UncheckedIOException (ex); }
} }
}
private void getPdfValidationResults(StringBuilder finalStringResult, PDFValidator pdfv, XMLValidator xv) throws IrrecoverableValidationError { private void getPdfValidationResults(StringBuilder finalStringResult, PDFValidator pdfv, XMLValidator xv) throws IrrecoverableValidationError {
finalStringResult.append(pdfv.getXMLResult()); finalStringResult.append(pdfv.getXMLResult());