diff --git a/codepropertygraph/src/main/scala/io/shiftleft/utils/IOUtils.scala b/codepropertygraph/src/main/scala/io/shiftleft/utils/IOUtils.scala index 581eefe8a..e10f2a318 100644 --- a/codepropertygraph/src/main/scala/io/shiftleft/utils/IOUtils.scala +++ b/codepropertygraph/src/main/scala/io/shiftleft/utils/IOUtils.scala @@ -1,21 +1,13 @@ package io.shiftleft.utils import java.io.Reader -import java.nio.charset.{CharsetDecoder, CodingErrorAction} -import java.nio.file.Path -import java.util.regex.Pattern +import java.nio.charset.{CharsetDecoder, CodingErrorAction, StandardCharsets} +import java.nio.file.{Files, Path} import scala.io.{BufferedSource, Codec, Source} -import scala.jdk.CollectionConverters.IteratorHasAsScala import scala.util.Using object IOUtils { - private val boms: Set[Char] = Set( - '\uefbb', // UTF-8 - '\ufeff', // UTF-16 (BE) - '\ufffe' // UTF-16 (LE) - ) - /** Creates a new UTF-8 decoder. Sadly, instances of CharsetDecoder are not thread-safe as the doc states: 'Instances * of this class are not safe for use by multiple concurrent threads.' (copied from: * [[java.nio.charset.CharsetDecoder]]) @@ -28,11 +20,12 @@ object IOUtils { .onMalformedInput(CodingErrorAction.REPLACE) .onUnmappableCharacter(CodingErrorAction.REPLACE) + /** Skips a leading byte order mark (BOM) if present. Once decoded as UTF-8, a BOM shows up as the single character + * U+FEFF (e.g. the UTF-8 BOM bytes EF BB BF decode to U+FEFF). + */ private def skipBOMIfPresent(reader: Reader): Unit = { reader.mark(1) - val possibleBOM = new Array[Char](1) - reader.read(possibleBOM) - if (!boms.contains(possibleBOM(0))) { + if (reader.read() != '\ufeff') { reader.reset() } } @@ -40,31 +33,17 @@ object IOUtils { private def contentFromBufferedSource(bufferedSource: BufferedSource): Seq[String] = { val reader = bufferedSource.bufferedReader() skipBOMIfPresent(reader) - reader.lines().iterator().asScala.toSeq - } - - private def contentStringFromBufferedSource(bufferedSource: BufferedSource): String = { - val reader = bufferedSource.bufferedReader() - val stringBuilder = new StringBuilder - val bufferSize = 1024 - var productive = true - - skipBOMIfPresent(reader) - while (productive) { - val buffer = new Array[Char](bufferSize) - val read = reader.read(buffer) - productive = read > 0 - if (productive) { - stringBuilder.appendAll(buffer, 0, read) - } + val lines = List.newBuilder[String] + var line = reader.readLine() + while (line != null) { + lines += line + line = reader.readLine() } - - stringBuilder.toString + lines.result() } /** Reads a file at the given path and: * - skips BOM if present - * - removes unpaired surrogates * - uses UTF-8 encoding (replacing malformed and unmappable characters) * * @param path @@ -77,7 +56,6 @@ object IOUtils { /** Reads a file at the given path and: * - skips BOM if present - * - removes unpaired surrogates * - uses UTF-8 encoding (replacing malformed and unmappable characters) * * @param path @@ -85,7 +63,18 @@ object IOUtils { * @return * a String with the given file's contents */ - def readEntireFile(path: Path): String = - Using.resource(Source.fromFile(path.toFile)(createDecoder()))(contentStringFromBufferedSource) + def readEntireFile(path: Path): String = { + val bytes = Files.readAllBytes(path) + // Strip the UTF-8 BOM (EF BB BF) if present. The String constructor replaces malformed and unmappable input with + // the default replacement character, i.e. decoding is lenient just like in readLinesInFile above. + val offset = if (startsWithUtf8Bom(bytes)) Utf8BomLength else 0 + new String(bytes, offset, bytes.length - offset, StandardCharsets.UTF_8) + } + + private val Utf8BomLength = 3 + + private def startsWithUtf8Bom(bytes: Array[Byte]): Boolean = + bytes.length >= Utf8BomLength && + bytes(0) == 0xef.toByte && bytes(1) == 0xbb.toByte && bytes(2) == 0xbf.toByte } diff --git a/codepropertygraph/src/test/scala/io/shiftleft/utils/IOUtilsTest.scala b/codepropertygraph/src/test/scala/io/shiftleft/utils/IOUtilsTest.scala new file mode 100644 index 000000000..33e3216e3 --- /dev/null +++ b/codepropertygraph/src/test/scala/io/shiftleft/utils/IOUtilsTest.scala @@ -0,0 +1,92 @@ +package io.shiftleft.utils + +import org.scalatest.matchers.should.Matchers +import org.scalatest.wordspec.AnyWordSpec + +import java.nio.charset.StandardCharsets +import java.nio.file.{Files, Path} + +class IOUtilsTest extends AnyWordSpec with Matchers { + + private val Utf8Bom = Array(0xef.toByte, 0xbb.toByte, 0xbf.toByte) + + private def withTempFile(content: Array[Byte])(f: Path => Unit): Unit = { + val path = Files.createTempFile("io-utils-test", ".tmp") + try { + Files.write(path, content) + f(path) + } finally { + Files.deleteIfExists(path) + } + } + + private def utf8(s: String): Array[Byte] = s.getBytes(StandardCharsets.UTF_8) + + "IOUtils.readLinesInFile" should { + "read all lines" in { + withTempFile(utf8("foo\nbar\r\nbaz")) { path => + IOUtils.readLinesInFile(path) shouldBe Seq("foo", "bar", "baz") + } + } + + "skip a UTF-8 BOM if present" in { + withTempFile(Utf8Bom ++ utf8("foo\nbar")) { path => + IOUtils.readLinesInFile(path) shouldBe Seq("foo", "bar") + } + } + + "replace malformed UTF-8 input instead of throwing" in { + // 0xC3 starts a two-byte sequence but 0x28 ('(') is not a valid continuation byte + withTempFile(utf8("foo") ++ Array(0xc3.toByte, 0x28.toByte)) { path => + IOUtils.readLinesInFile(path) shouldBe Seq("foo\ufffd(") + } + } + + "not strip a leading character that merely encodes like a BOM" in { + // EF AE BB is the genuine UTF-8 encoding of U+FBBB (a regular character, not a BOM) + withTempFile(Array(0xef.toByte, 0xae.toByte, 0xbb.toByte)) { path => + IOUtils.readLinesInFile(path) shouldBe Seq("\ufbbb") + } + } + + "return an empty Seq for an empty file" in { + withTempFile(Array.emptyByteArray) { path => + IOUtils.readLinesInFile(path) shouldBe empty + } + } + } + + "IOUtils.readEntireFile" should { + "read the entire content" in { + withTempFile(utf8("foo\nbar\n")) { path => + IOUtils.readEntireFile(path) shouldBe "foo\nbar\n" + } + } + + "skip a UTF-8 BOM if present" in { + withTempFile(Utf8Bom ++ utf8("foo")) { path => + IOUtils.readEntireFile(path) shouldBe "foo" + } + } + + "replace malformed UTF-8 input instead of throwing" in { + withTempFile(Array(0xc3.toByte, 0x28.toByte)) { path => + IOUtils.readEntireFile(path) shouldBe "\ufffd(" + } + } + + "not strip a leading character that merely encodes like a BOM" in { + // EF AE BB is the genuine UTF-8 encoding of U+FBBB (a regular character, not a BOM) + withTempFile(Array(0xef.toByte, 0xae.toByte, 0xbb.toByte)) { path => + IOUtils.readEntireFile(path) shouldBe "\ufbbb" + } + } + + "return an empty String for an empty file" in { + withTempFile(Array.emptyByteArray) { path => + IOUtils.readEntireFile(path) shouldBe "" + } + } + } + +}