Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,7 @@ val df = spark.read
.option("header", "true") // Required
.option("treatEmptyValuesAsNulls", "false") // Optional, default: true
.option("setErrorCellsToFallbackValues", "true") // Optional, default: false, where errors will be converted to null. If true, any ERROR cell values (e.g. #N/A) will be converted to the zero values of the column's data type.
.option("usePlainNumberFormat", "false") // Optional, default: false, If true, format the cells without rounding and scientific notations
.option("usePlainNumberFormat", "false") // Optional, default: false. "true": format General/@-formatted cells without rounding and scientific notations; "all": every non-date numeric cell, regardless of its number format
.option("inferSchema", "false") // Optional, default: false
.option("addColorColumns", "true") // Optional, default: false
.option("timestampFormat", "MM-dd-yyyy HH:mm:ss") // Optional, default: yyyy-mm-dd hh:mm:ss[.fffffffff]
Expand All @@ -95,7 +95,7 @@ val df = spark.read.excel(
dataAddress = "'My Sheet'!B3:C35", // Optional, default: "A1"
treatEmptyValuesAsNulls = false, // Optional, default: true
setErrorCellsToFallbackValues = false, // Optional, default: false, where errors will be converted to null. If true, any ERROR cell values (e.g. #N/A) will be converted to the zero values of the column's data type.
usePlainNumberFormat = false, // Optional, default: false. If true, format the cells without rounding and scientific notations
usePlainNumberFormat = false, // Optional, default: false. true: format General/@-formatted cells without rounding and scientific notations; PlainNumberFormatMode.All: every non-date numeric cell, regardless of its number format
inferSchema = false, // Optional, default: false
addColorColumns = true, // Optional, default: false
timestampFormat = "MM-dd-yyyy HH:mm:ss", // Optional, default: yyyy-mm-dd hh:mm:ss[.fffffffff]
Expand Down
21 changes: 19 additions & 2 deletions src/main/scala/dev/mauch/spark/excel/DataColumn.scala
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ import org.apache.spark.sql.types._

import java.math.BigDecimal
import java.sql.{Date, Timestamp}
import java.text.FieldPosition
import scala.util.{Failure, Success, Try}

trait DataColumn extends PartialFunction[Seq[Cell], Any] {
Expand All @@ -35,20 +36,34 @@ class HeaderDataColumn(
val field: StructField,
val columnIndex: Int,
treatEmptyValuesAsNulls: Boolean,
usePlainNumberFormat: Boolean,
usePlainNumberFormat: PlainNumberFormatMode,
parseTimestamp: String => Timestamp,
parseDate: String => Date,
setErrorCellsToFallbackValues: Boolean
) extends DataColumn {
def name: String = field.name

/** Whether this (cached-)numeric cell should be rendered at full precision, ignoring its number format. Date cells
* keep their formatted rendering, non-finite values keep POI's display rendering.
*/
private def renderPlainNumber(cell: Cell): Boolean =
usePlainNumberFormat == PlainNumberFormatMode.All && !DateUtil.isCellDateFormatted(cell) &&
java.lang.Double.isFinite(cell.getNumericCellValue)

/** Same invocation POI's DataFormatter uses for a registered format, so digits match usePlainNumberFormat=true's. */
private def plainNumberString(cell: Cell): String =
PlainNumberFormat
.format(BigDecimal.valueOf(cell.getNumericCellValue), new StringBuffer(), new FieldPosition(0))
.toString

def extractValue(cell: Cell): Any = {
val cellType = if (cell.getCellType == CellType.FORMULA) cell.getCachedFormulaResultType else cell.getCellType
if (cellType == CellType.BLANK) {
return null
}

lazy val dataFormatter = new DataFormatter()
if (usePlainNumberFormat) {
if (usePlainNumberFormat != PlainNumberFormatMode.Off) {
// Overwrite ExcelGeneralNumberFormat with custom PlainNumberFormat.
// See https://github.dev/mauch/spark-excel/issues/321
lazy val plainNumberFormat = PlainNumberFormat
Expand All @@ -65,11 +80,13 @@ class HeaderDataColumn(
case CellType.FORMULA =>
cell.getCachedFormulaResultType match {
case CellType.STRING => Option(cell.getRichStringCellValue).map(_.getString)
case CellType.NUMERIC if renderPlainNumber(cell) => Some(plainNumberString(cell))
case CellType.NUMERIC => Option(cell.getNumericCellValue).map(_.toString)
case CellType.BLANK => None
case _ => Some(dataFormatter.formatCellValue(cell))
}
case CellType.BLANK => None
case CellType.NUMERIC if renderPlainNumber(cell) => Some(plainNumberString(cell))
case _ => Some(dataFormatter.formatCellValue(cell))
}
def parseNumber(string: Option[String]): Option[Double] = string.filter(_.trim.nonEmpty).map(stringToDouble)
Expand Down
2 changes: 1 addition & 1 deletion src/main/scala/dev/mauch/spark/excel/DefaultSource.scala
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,7 @@ class DefaultSource extends RelationProvider with SchemaRelationProvider with Cr
header = checkParameter(parameters, "header").toBoolean,
treatEmptyValuesAsNulls = parameters.get("treatEmptyValuesAsNulls").fold(false)(_.toBoolean),
setErrorCellsToFallbackValues = parameters.get("setErrorCellsToFallbackValues").fold(false)(_.toBoolean),
usePlainNumberFormat = parameters.get("usePlainNumberFormat").fold(false)(_.toBoolean),
usePlainNumberFormat = PlainNumberFormatMode.parse(parameters.getOrElse("usePlainNumberFormat", "false")),
userSchema = Option(schema),
inferSheetSchema = parameters.get("inferSchema").fold(false)(_.toBoolean),
addColorColumns = parameters.get("addColorColumns").fold(false)(_.toBoolean),
Expand Down
2 changes: 1 addition & 1 deletion src/main/scala/dev/mauch/spark/excel/ExcelRelation.scala
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@ case class ExcelRelation(
dataLocator: DataLocator,
header: Boolean,
treatEmptyValuesAsNulls: Boolean,
usePlainNumberFormat: Boolean,
usePlainNumberFormat: PlainNumberFormatMode,
inferSheetSchema: Boolean,
setErrorCellsToFallbackValues: Boolean,
addColorColumns: Boolean = true,
Expand Down
5 changes: 3 additions & 2 deletions src/main/scala/dev/mauch/spark/excel/PlainNumberFormat.scala
Original file line number Diff line number Diff line change
Expand Up @@ -38,8 +38,9 @@ object PlainNumberFormat extends Format {
// It's an integer, format without decimal point
toAppendTo.append(stripped.toBigInteger().toString())
} else {
// It's not an integer, format as plain string
toAppendTo.append(bd.toPlainString)
// It's not an integer, format the stripped value so no trailing zero from
// Double.toString's "d.0E-x" mantissa survives (0.0005 must not become "0.00050")
toAppendTo.append(stripped.toPlainString)
}
}

Expand Down
57 changes: 57 additions & 0 deletions src/main/scala/dev/mauch/spark/excel/PlainNumberFormatMode.scala
Original file line number Diff line number Diff line change
@@ -0,0 +1,57 @@
/*
* Copyright 2022 Martin Mauch (@nightscape)
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package dev.mauch.spark.excel

import java.util.Locale
import scala.language.implicitConversions

/** Which numeric cells the `usePlainNumberFormat` read option renders through [[PlainNumberFormat]], i.e. at full
* precision without rounding or scientific notation. Spelled `false`, `true` or `all` in the option.
*/
sealed trait PlainNumberFormatMode {
def optionValue: String
}

object PlainNumberFormatMode {

/** Cells render as POI displays them, through their number format. */
case object Off extends PlainNumberFormatMode { val optionValue = "false" }

/** Cells whose number format is `General` or `@` render through [[PlainNumberFormat]]; cells with an explicit number
* format keep their formatted rendering.
*/
case object General extends PlainNumberFormatMode { val optionValue = "true" }

/** Every non-date numeric cell renders through [[PlainNumberFormat]] regardless of its number format; date-formatted
* cells keep their formatted rendering.
*/
case object All extends PlainNumberFormatMode { val optionValue = "all" }

val values: Seq[PlainNumberFormatMode] = Seq(Off, General, All)

def parse(value: String): PlainNumberFormatMode = {
val normalized = Option(value).map(_.toLowerCase(Locale.ROOT)).getOrElse(Off.optionValue)
values.find(_.optionValue == normalized).getOrElse {
throw new IllegalArgumentException(
s"usePlainNumberFormat must be one of ${values.map(_.optionValue).mkString(", ")}, got '$value'"
)
}
}

/** Keeps `usePlainNumberFormat = true` / `= false` compiling in the Scala API. */
implicit def fromBoolean(enabled: Boolean): PlainNumberFormatMode = if (enabled) General else Off
}
4 changes: 2 additions & 2 deletions src/main/scala/dev/mauch/spark/excel/package.scala
Original file line number Diff line number Diff line change
Expand Up @@ -77,7 +77,7 @@ package object excel {
treatEmptyValuesAsNulls: Boolean = false,
setErrorCellsToFallbackValues: Boolean = false,
inferSchema: Boolean = false,
usePlainNumberFormat: Boolean = false,
usePlainNumberFormat: PlainNumberFormatMode = PlainNumberFormatMode.Off,
addColorColumns: Boolean = false,
dataAddress: String = null,
timestampFormat: String = null,
Expand All @@ -91,7 +91,7 @@ package object excel {
"header" -> header,
"treatEmptyValuesAsNulls" -> treatEmptyValuesAsNulls,
"setErrorCellsToFallbackValues" -> setErrorCellsToFallbackValues,
"usePlainNumberFormat" -> usePlainNumberFormat,
"usePlainNumberFormat" -> usePlainNumberFormat.optionValue,
"inferSchema" -> inferSchema,
"addColorColumns" -> addColorColumns,
"dataAddress" -> dataAddress,
Expand Down
4 changes: 3 additions & 1 deletion src/main/scala/dev/mauch/spark/excel/v2/ExcelGenerator.scala
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@

package dev.mauch.spark.excel.v2

import dev.mauch.spark.excel.PlainNumberFormatMode
import org.apache.hadoop.conf.Configuration
import org.apache.hadoop.fs.Path
import org.apache.poi.hssf.usermodel.HSSFWorkbook
Expand Down Expand Up @@ -103,7 +104,8 @@ class ExcelGenerator(val path: String, val dataSchema: StructType, val conf: Con
private lazy val TimestampCellStyle = createStyle(options.timestampFormat)
private lazy val WholeNumberCellStyle = createStyle("General")
private lazy val DecimalNumberCellStyle =
if (options.usePlainNumberFormat) createStyle("General") else createStyle("0.00E+000")
if (options.usePlainNumberFormat != PlainNumberFormatMode.Off) createStyle("General")
else createStyle("0.00E+000")
private lazy val StringCellStyle = createStyle("@")

private def makeConverter(dataType: DataType): ValueConverter = dataType match {
Expand Down
39 changes: 33 additions & 6 deletions src/main/scala/dev/mauch/spark/excel/v2/ExcelHelper.scala
Original file line number Diff line number Diff line change
Expand Up @@ -17,12 +17,13 @@
package dev.mauch.spark.excel.v2

import com.github.pjfanning.xlsx.StreamingReader
import dev.mauch.spark.excel.PlainNumberFormatMode
import org.apache.hadoop.conf.Configuration
import org.apache.hadoop.fs.{FileSystem, Path}
import org.apache.poi.hssf.usermodel.HSSFWorkbookFactory
import org.apache.poi.openxml4j.util.ZipInputStreamZipEntrySource
import org.apache.poi.ss.SpreadsheetVersion
import org.apache.poi.ss.usermodel.{Cell, CellType, DataFormatter, FormulaError, Workbook, WorkbookFactory}
import org.apache.poi.ss.usermodel.{Cell, CellType, DataFormatter, DateUtil, FormulaError, Workbook, WorkbookFactory}
import org.apache.poi.ss.util.{AreaReference, CellReference}
import org.apache.poi.util.IOUtils
import org.apache.poi.xssf.usermodel.XSSFWorkbookFactory
Expand Down Expand Up @@ -52,8 +53,9 @@ object PlainNumberFormat extends Format {
// It's an integer, format without decimal point
toAppendTo.append(stripped.toBigInteger().toString())
} else {
// It's not an integer, format as plain string
toAppendTo.append(bd.toPlainString)
// It's not an integer, format the stripped value so no trailing zero from
// Double.toString's "d.0E-x" mantissa survives (0.0005 must not become "0.00050")
toAppendTo.append(stripped.toPlainString)
}
}

Expand All @@ -67,7 +69,7 @@ class ExcelHelper private (options: ExcelOptions) {
/* For get cell string value */
private lazy val dataFormatter = {
val r = new DataFormatter()
if (options.usePlainNumberFormat) {
if (options.usePlainNumberFormat != PlainNumberFormatMode.Off) {

/* Overwrite ExcelGeneralNumberFormat with custom PlainNumberFormat. See
* https://github.dev/mauch/spark-excel/issues/321
Expand Down Expand Up @@ -97,14 +99,39 @@ class ExcelHelper private (options: ExcelOptions) {
*/
case CellType.ERROR => FormulaError.forInt(cell.getErrorCellValue).getString
case CellType.STRING => cell.getStringCellValue
case CellType.NUMERIC if renderPlainNumber(cell) => plainNumberString(cell)
case CellType.NUMERIC => cell.getNumericCellValue.toString

/* Get what displayed on the cell, for all other cases */
case _ => dataFormatter.formatCellValue(cell)
}
case CellType.NUMERIC if renderPlainNumber(cell) => plainNumberString(cell)
case _ => dataFormatter.formatCellValue(cell)
}

/** Whether this (cached-)numeric cell should be rendered at full precision, ignoring its number format. Date cells
* keep their formatted rendering, non-finite values keep POI's display rendering.
*/
private def renderPlainNumber(cell: Cell): Boolean =
options.usePlainNumberFormat == PlainNumberFormatMode.All && !DateUtil.isCellDateFormatted(cell) &&
java.lang.Double.isFinite(cell.getNumericCellValue)

/** Render a numeric cell at full precision, ignoring its number format. Invokes [[PlainNumberFormat]] with the same
* argument POI's DataFormatter passes to a registered format, so the rendering is identical to
* usePlainNumberFormat=true's for General-format cells.
*/
private def plainNumberString(cell: Cell): String =
PlainNumberFormat
.format(BigDecimal.valueOf(cell.getNumericCellValue), new StringBuffer(), new FieldPosition(0))
.toString

/** Header-cell rendering, honoring usePlainNumberFormat=all so a numeric header is named consistently with its data
* cells.
*/
private def headerCellString(cell: Cell): String =
if (cell.getCellType == CellType.NUMERIC && renderPlainNumber(cell)) plainNumberString(cell)
else dataFormatter.formatCellValue(cell)

/** Get workbook
*
* @param conf
Expand Down Expand Up @@ -230,14 +257,14 @@ class ExcelHelper private (options: ExcelOptions) {

val dataColumns =
if (options.header) {
val headerNames = firstRow.map(dataFormatter.formatCellValue)
val headerNames = firstRow.map(headerCellString)
val duplicates = {
val nonNullHeaderNames = headerNames.filter(_ != null)
nonNullHeaderNames.groupBy(identity).filter(_._2.size > 1).keySet
}

firstRow.zipWithIndex.map { case (cell, index) =>
val value = dataFormatter.formatCellValue(cell)
val value = headerCellString(cell)
val cellType = cell.getCellType
if (
cellType == CellType.ERROR || cellType == CellType.BLANK ||
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@

package dev.mauch.spark.excel.v2

import dev.mauch.spark.excel.PlainNumberFormatMode
import org.apache.spark.sql.catalyst.util.{
CaseInsensitiveMap,
DateFormatter,
Expand Down Expand Up @@ -103,8 +104,10 @@ trait ExcelOptionsTrait extends Serializable {
val positiveInf = parameters.getOrElse("positiveInf", "Inf")
val negativeInf = parameters.getOrElse("negativeInf", "-Inf")

/* If true, format the cells without rounding and scientific notations */
val usePlainNumberFormat = getBool("usePlainNumberFormat", default = false)
/* Which numeric cells to format without rounding and scientific notations: false (none), true (General/@-formatted
cells), all (every non-date numeric cell, ignoring its number format) */
val usePlainNumberFormat: PlainNumberFormatMode =
PlainNumberFormatMode.parse(parameters.getOrElse("usePlainNumberFormat", "false"))

/* If true, keep undefined (Excel) rows */
val keepUndefinedRows = getBool("keepUndefinedRows", default = false)
Expand Down
Binary file not shown.
Binary file not shown.
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
/*
* Copyright 2022 Martin Mauch (@nightscape)
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package dev.mauch.spark.excel

import org.scalatest.funsuite.AnyFunSuite

class PlainNumberFormatModeSuite extends AnyFunSuite {

test("parses false, true and all case-insensitively, absent as false") {
assert(PlainNumberFormatMode.parse("false") == PlainNumberFormatMode.Off)
assert(PlainNumberFormatMode.parse("TRUE") == PlainNumberFormatMode.General)
assert(PlainNumberFormatMode.parse("ALL") == PlainNumberFormatMode.All)
assert(PlainNumberFormatMode.parse(null) == PlainNumberFormatMode.Off)
}

test("rejects any other value, naming the option and the value") {
val e = intercept[IllegalArgumentException](PlainNumberFormatMode.parse("yes"))
assert(e.getMessage.contains("usePlainNumberFormat") && e.getMessage.contains("yes"))
}
}
Loading