Extract CsvReader into separate ebean-csv-reader module

This commit is contained in:
Rob Bygrave
2022-08-25 17:25:49 +12:00
parent 5b2a8cdacb
commit 216c43eea6
22 changed files with 372 additions and 130 deletions
@@ -0,0 +1,249 @@
package io.ebeaninternal.server.text.csv;
// Original name: au.com.bytecode.opencsv.CSVReader
// rbygrave: Made some Java Generics tweaks to remove warnings
/**
* Copyright 2005 Bytecode Pty Ltd.
* <p>
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
* <p>
* http://www.apache.org/licenses/LICENSE-2.0
* <p>
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
import java.io.BufferedReader;
import java.io.IOException;
import java.io.Reader;
import java.util.ArrayList;
import java.util.List;
/**
* Glen Smith's CSV reader released under Apache License version 2.
*
* @author Glen Smith
*
*/
public class CsvUtilReader {
private final BufferedReader br;
private boolean hasNext = true;
private final char separator;
private final char quotechar;
private final int skipLines;
private boolean linesSkiped;
/** The default separator to use if none is supplied to the constructor. */
public static final char DEFAULT_SEPARATOR = ',';
/**
* The default quote character to use if none is supplied to the
* constructor.
*/
public static final char DEFAULT_QUOTE_CHARACTER = '"';
/**
* The default line to start reading.
*/
public static final int DEFAULT_SKIP_LINES = 0;
/**
* Constructs CSVReader using a comma for the separator.
*
* @param reader
* the reader to an underlying CSV source.
*/
public CsvUtilReader(Reader reader) {
this(reader, DEFAULT_SEPARATOR);
}
/**
* Constructs CSVReader with supplied separator.
*
* @param reader
* the reader to an underlying CSV source.
* @param separator
* the delimiter to use for separating entries.
*/
public CsvUtilReader(Reader reader, char separator) {
this(reader, separator, DEFAULT_QUOTE_CHARACTER);
}
/**
* Constructs CSVReader with supplied separator and quote char.
*
* @param reader
* the reader to an underlying CSV source.
* @param separator
* the delimiter to use for separating entries
* @param quotechar
* the character to use for quoted elements
*/
public CsvUtilReader(Reader reader, char separator, char quotechar) {
this(reader, separator, quotechar, DEFAULT_SKIP_LINES);
}
/**
* Constructs CSVReader with supplied separator and quote char.
*
* @param reader
* the reader to an underlying CSV source.
* @param separator
* the delimiter to use for separating entries
* @param quotechar
* the character to use for quoted elements
* @param line
* the line number to skip for start reading
*/
public CsvUtilReader(Reader reader, char separator, char quotechar, int line) {
this.br = new BufferedReader(reader);
this.separator = separator;
this.quotechar = quotechar;
this.skipLines = line;
}
/**
* Reads the entire file into a List with each element being a String[] of
* tokens.
*
* @return a List of String[], with each String[] representing a line of the
* file.
*
* @throws IOException
* if bad things happen during the read
*/
public List<String[]> readAll() throws IOException {
List<String[]> allElements = new ArrayList<>();
while (hasNext) {
String[] nextLineAsTokens = readNext();
if (nextLineAsTokens != null) {
allElements.add(nextLineAsTokens);
}
}
return allElements;
}
/**
* Reads the next line from the buffer and converts to a string array.
*
* @return a string array with each comma-separated element as a separate
* entry.
*
* @throws IOException
* if bad things happen during the read
*/
public String[] readNext() throws IOException {
String nextLine = getNextLine();
return hasNext ? parseLine(nextLine) : null;
}
/**
* Reads the next line from the file.
*
* @return the next line from the file without trailing newline
* @throws IOException
* if bad things happen during the read
*/
private String getNextLine() throws IOException {
if (!this.linesSkiped) {
for (int i = 0; i < skipLines; i++) {
br.readLine();
}
this.linesSkiped = true;
}
String nextLine = br.readLine();
if (nextLine == null) {
hasNext = false;
}
return hasNext ? nextLine : null;
}
/**
* Parses an incoming String and returns an array of elements.
*
* @param nextLine
* the string to parse
* @return the comma-tokenized list of elements, or null if nextLine is null
* @throws IOException if bad things happen during the read
*/
private String[] parseLine(String nextLine) throws IOException {
if (nextLine == null) {
return null;
}
List<String> tokensOnThisLine = new ArrayList<>();
StringBuilder sb = new StringBuilder();
boolean inQuotes = false;
do {
if (inQuotes) {
// continuing a quoted section, reappend newline
sb.append("\n");
nextLine = getNextLine();
if (nextLine == null)
break;
}
for (int i = 0; i < nextLine.length(); i++) {
char c = nextLine.charAt(i);
if (c == quotechar) {
// this gets complex... the quote may end a quoted block, or escape another quote.
// do a 1-char lookahead:
if (inQuotes // we are in quotes, therefore there can be escaped quotes in here.
&& nextLine.length() > (i + 1) // there is indeed another character to check.
&& nextLine.charAt(i + 1) == quotechar) { // ..and that char. is a quote also.
// we have two quote chars in a row == one quote char, so consume them both and
// put one on the token. we do *not* exit the quoted text.
sb.append(nextLine.charAt(i + 1));
i++;
} else {
inQuotes = !inQuotes;
// the tricky case of an embedded quote in the middle: a,bc"d"ef,g
if (i > 2 //not on the begining of the line
&& nextLine.charAt(i - 1) != this.separator //not at the begining of an escape sequence
&& nextLine.length() > (i + 1) &&
nextLine.charAt(i + 1) != this.separator //not at the end of an escape sequence
) {
sb.append(c);
}
}
} else if (c == separator && !inQuotes) {
tokensOnThisLine.add(sb.toString().trim());
sb = new StringBuilder(); // start work on next token
} else {
sb.append(c);
}
}
} while (inQuotes);
tokensOnThisLine.add(sb.toString().trim());
return tokensOnThisLine.toArray(new String[0]);
}
/**
* Closes the underlying reader.
*
* @throws IOException if the close fails
*/
public void close() throws IOException {
br.close();
}
}
@@ -0,0 +1,352 @@
package io.ebeaninternal.server.text.csv;
import io.ebean.Database;
import io.ebean.bean.EntityBean;
import io.ebean.plugin.BeanType;
import io.ebean.plugin.ExpressionPath;
import io.ebean.text.StringParser;
import io.ebean.text.TextException;
import io.ebean.text.csv.CsvCallback;
import io.ebean.text.csv.CsvReader;
import io.ebean.text.csv.DefaultCsvCallback;
import java.io.Reader;
import java.sql.Types;
import java.text.DateFormat;
import java.text.ParseException;
import java.text.SimpleDateFormat;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Date;
import java.util.List;
import java.util.Locale;
/**
* Implementation of the CsvReader
*/
public class TCsvReader<T> {//implements CsvReader<T> {
private static final TimeStringParser TIME_PARSER = new TimeStringParser();
private final Database server;
private final BeanType<T> descriptor;
private final List<CsvColumn> columnList = new ArrayList<>();
private final CsvColumn ignoreColumn = new CsvColumn();
private boolean hasHeader;
private int logInfoFrequency = 1000;
private String defaultTimeFormat = "HH:mm:ss";
private String defaultDateFormat = "yyyy-MM-dd";
private String defaultTimestampFormat = "yyyy-MM-dd hh:mm:ss.fffffffff";
private Locale defaultLocale = Locale.getDefault();
/**
* The batch size used for JDBC statement batching.
*/
protected int persistBatchSize = 30;
private boolean addPropertiesFromHeader;
public TCsvReader(Database server, BeanType<T> descriptor) {
this.server = server;
this.descriptor = descriptor;
}
////@Override
public void setDefaultLocale(Locale defaultLocale) {
this.defaultLocale = defaultLocale;
}
////@Override
public void setDefaultTimeFormat(String defaultTimeFormat) {
this.defaultTimeFormat = defaultTimeFormat;
}
////@Override
public void setDefaultDateFormat(String defaultDateFormat) {
this.defaultDateFormat = defaultDateFormat;
}
////@Override
public void setDefaultTimestampFormat(String defaultTimestampFormat) {
this.defaultTimestampFormat = defaultTimestampFormat;
}
////@Override
public void setPersistBatchSize(int persistBatchSize) {
this.persistBatchSize = persistBatchSize;
}
////@Override
public void setIgnoreHeader() {
setHasHeader(true, false);
}
////@Override
public void setAddPropertiesFromHeader() {
setHasHeader(true, true);
}
//@Override
public void setHasHeader(boolean hasHeader, boolean addPropertiesFromHeader) {
this.hasHeader = hasHeader;
this.addPropertiesFromHeader = addPropertiesFromHeader;
}
//@Override
public void setLogInfoFrequency(int logInfoFrequency) {
this.logInfoFrequency = logInfoFrequency;
}
//@Override
public void addIgnore() {
columnList.add(ignoreColumn);
}
//@Override
public void addProperty(String propertyName) {
addProperty(propertyName, null);
}
//@Override
public void addDateTime(String propertyName, String dateTimeFormat) {
addDateTime(propertyName, dateTimeFormat, Locale.getDefault());
}
//@Override
public void addDateTime(String propertyName, String dateTimeFormat, Locale locale) {
ExpressionPath elProp = descriptor.expressionPath(propertyName);
if (dateTimeFormat == null) {
dateTimeFormat = getDefaultDateTimeFormat(elProp.jdbcType());
}
if (locale == null) {
locale = defaultLocale;
}
SimpleDateFormat sdf = new SimpleDateFormat(dateTimeFormat, locale);
DateTimeParser parser = new DateTimeParser(sdf, dateTimeFormat, elProp);
CsvColumn column = new CsvColumn(elProp, parser);
columnList.add(column);
}
private String getDefaultDateTimeFormat(int jdbcType) {
switch (jdbcType) {
case Types.TIME:
return defaultTimeFormat;
case Types.DATE:
return defaultDateFormat;
case Types.TIMESTAMP:
return defaultTimestampFormat;
default:
throw new RuntimeException("Expected java.sql.Types TIME,DATE or TIMESTAMP but got [" + jdbcType + "]");
}
}
//@Override
public void addProperty(String propertyName, StringParser parser) {
ExpressionPath elProp = descriptor.expressionPath(propertyName);
if (parser == null) {
parser = elProp.stringParser();
}
CsvColumn column = new CsvColumn(elProp, parser);
columnList.add(column);
}
//@Override
public void process(Reader reader) throws Exception {
DefaultCsvCallback<T> callback = new DefaultCsvCallback<>(persistBatchSize, logInfoFrequency);
process(reader, callback);
}
//@Override
public void process(Reader reader, CsvCallback<T> callback) throws Exception {
if (reader == null) {
throw new NullPointerException("reader is null?");
}
if (callback == null) {
throw new NullPointerException("callback is null?");
}
CsvUtilReader utilReader = new CsvUtilReader(reader);
callback.begin(server);
int row = 0;
if (hasHeader) {
String[] line = utilReader.readNext();
if (addPropertiesFromHeader) {
addPropertiesFromHeader(line);
}
callback.readHeader(line);
}
try {
do {
++row;
String[] line = utilReader.readNext();
if (line == null) {
--row;
break;
}
if (callback.processLine(row, line)) {
// the line content is expected to be ok for processing
if (line.length != columnList.size()) {
// we have not got the expected number of columns
String msg = "Error at line " + row + ". Expected [" + columnList.size() + "] columns "
+ "but instead we have [" + line.length + "]. Line[" + Arrays.toString(line) + "]";
throw new TextException(msg);
}
T bean = buildBeanFromLineContent(row, line);
callback.processBean(row, line, bean);
}
} while (true);
callback.end(row);
} catch (Exception e) {
// notify that an error occurred so that any
// transaction can be rolled back if required
callback.endWithError(row, e);
throw e;
}
}
private void addPropertiesFromHeader(String[] line) {
for (String aLine : line) {
ExpressionPath elProp = descriptor.expressionPath(aLine);
//ElPropertyValue elProp = descriptor.elGetValue(aLine);
if (elProp == null) {
throw new TextException("Property [" + aLine + "] not found");
}
if (Types.TIME == elProp.jdbcType()) {
addProperty(aLine, TIME_PARSER);
} else if (isDateTimeType(elProp.jdbcType())) {
addDateTime(aLine, null, null);
// } else if (elProp.isAssocProperty()) {
// BeanPropertyAssocOne<?> assocOne = (BeanPropertyAssocOne<?>) elProp.beanProperty();
// String idProp = assocOne.descriptor().idBinder().getIdProperty();
// addProperty(aLine + "." + idProp);
} else {
addProperty(aLine);
}
}
}
private boolean isDateTimeType(int t) {
return t == Types.TIMESTAMP || t == Types.DATE || t == Types.TIME;
}
@SuppressWarnings("unchecked")
protected T buildBeanFromLineContent(int row, String[] line) {
try {
T bean = descriptor.createBean();
EntityBean entityBean = (EntityBean)bean;
for (int columnPos = 0; columnPos < line.length; columnPos++) {
convertAndSetColumn(columnPos, line[columnPos], entityBean);
}
return bean;
} catch (RuntimeException e) {
String msg = "Error at line: " + row + " line[" + Arrays.toString(line) + "]";
throw new RuntimeException(msg, e);
}
}
protected void convertAndSetColumn(int columnPos, String strValue, EntityBean bean) {
strValue = strValue.trim();
if (strValue.isEmpty()) {
return;
}
CsvColumn c = columnList.get(columnPos);
c.convertAndSet(strValue, bean);
}
/**
* Processes a column in the csv content.
*/
public static class CsvColumn {
private final ExpressionPath path;
private final StringParser parser;
/**
* Constructor for the IGNORE column.
*/
private CsvColumn() {
this.path = null;
this.parser = null;
}
/**
* Construct with a property and parser.
*/
public CsvColumn(ExpressionPath path, StringParser parser) {
this.path = path;
this.parser = parser;
}
/**
* Convert the string to the appropriate value and set it to the bean.
*/
public void convertAndSet(String strValue, EntityBean bean) {
if (parser != null && path != null) {
Object value = parser.parse(strValue);
path.pathSet(bean, value);
}
}
}
/**
* A StringParser for converting custom date/time/datetime strings into
* appropriate java types (Date, Calendar, SQL Date, Time, Timestamp, JODA
* etc).
*/
private static class DateTimeParser implements StringParser {
private final DateFormat dateFormat;
private final ExpressionPath path;
private final String format;
DateTimeParser(DateFormat dateFormat, String format, ExpressionPath path) {
this.dateFormat = dateFormat;
this.path = path;
this.format = format;
}
//@Override
public Object parse(String value) {
try {
Date dt = dateFormat.parse(value);
return path.parseDateTime(dt.getTime());
} catch (ParseException e) {
throw new TextException("Error parsing [{}] using format[" + format + "]", value, e);
}
}
}
}
@@ -0,0 +1,57 @@
package io.ebeaninternal.server.text.csv;
import io.ebean.text.StringParser;
import java.sql.Time;
/**
* Parser for TIME types that supports both HH:mm:ss and HH:mm.
*/
public final class TimeStringParser implements StringParser {
private static final TimeStringParser SHARED = new TimeStringParser();
/**
* Return a shared instance as this is thread safe.
*/
public static TimeStringParser get() {
return SHARED;
}
/**
* Parse the String supporting both HH:mm:ss and HH:mm formats.
*/
@Override
@SuppressWarnings("deprecation")
public Object parse(String value) {
if (value == null || value.trim().isEmpty()) {
return null;
}
String s = value.trim();
int firstColon = s.indexOf(':');
if (firstColon == -1) {
throw new java.lang.IllegalArgumentException("No ':' in value [" + s + "]");
}
try {
int second;
int minute;
int hour = Integer.parseInt(s.substring(0, firstColon));
int secondColon = s.indexOf(':', firstColon + 1);
if (secondColon == -1) {
minute = Integer.parseInt(s.substring(firstColon + 1, s.length()));
second = 0;
} else {
minute = Integer.parseInt(s.substring(firstColon + 1, secondColon));
second = Integer.parseInt(s.substring(secondColon + 1));
}
return new Time(hour, minute, second);
} catch (NumberFormatException e) {
throw new java.lang.IllegalArgumentException("Number format Error parsing time [" + s + "] " + e.getMessage(), e);
}
}
}