Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 25 additions & 36 deletions codepropertygraph/src/main/scala/io/shiftleft/utils/IOUtils.scala
Original file line number Diff line number Diff line change
@@ -1,21 +1,13 @@
package io.shiftleft.utils

import java.io.Reader
import java.nio.charset.{CharsetDecoder, CodingErrorAction}
import java.nio.file.Path
import java.util.regex.Pattern
import java.nio.charset.{CharsetDecoder, CodingErrorAction, StandardCharsets}
import java.nio.file.{Files, Path}
import scala.io.{BufferedSource, Codec, Source}
import scala.jdk.CollectionConverters.IteratorHasAsScala
import scala.util.Using

object IOUtils {

private val boms: Set[Char] = Set(
'\uefbb', // UTF-8
'\ufeff', // UTF-16 (BE)
'\ufffe' // UTF-16 (LE)
)

/** Creates a new UTF-8 decoder. Sadly, instances of CharsetDecoder are not thread-safe as the doc states: 'Instances
* of this class are not safe for use by multiple concurrent threads.' (copied from:
* [[java.nio.charset.CharsetDecoder]])
Expand All @@ -28,43 +20,30 @@ object IOUtils {
.onMalformedInput(CodingErrorAction.REPLACE)
.onUnmappableCharacter(CodingErrorAction.REPLACE)

/** Skips a leading byte order mark (BOM) if present. Once decoded as UTF-8, a BOM shows up as the single character
* U+FEFF (e.g. the UTF-8 BOM bytes EF BB BF decode to U+FEFF).
*/
private def skipBOMIfPresent(reader: Reader): Unit = {
reader.mark(1)
val possibleBOM = new Array[Char](1)
reader.read(possibleBOM)
if (!boms.contains(possibleBOM(0))) {
if (reader.read() != '\ufeff') {
reader.reset()
}
}

private def contentFromBufferedSource(bufferedSource: BufferedSource): Seq[String] = {
val reader = bufferedSource.bufferedReader()
skipBOMIfPresent(reader)
reader.lines().iterator().asScala.toSeq
}

private def contentStringFromBufferedSource(bufferedSource: BufferedSource): String = {
val reader = bufferedSource.bufferedReader()
val stringBuilder = new StringBuilder
val bufferSize = 1024
var productive = true

skipBOMIfPresent(reader)
while (productive) {
val buffer = new Array[Char](bufferSize)
val read = reader.read(buffer)
productive = read > 0
if (productive) {
stringBuilder.appendAll(buffer, 0, read)
}
val lines = List.newBuilder[String]
var line = reader.readLine()
while (line != null) {
lines += line
line = reader.readLine()
}

stringBuilder.toString
lines.result()
}

/** Reads a file at the given path and:
* - skips BOM if present
* - removes unpaired surrogates
* - uses UTF-8 encoding (replacing malformed and unmappable characters)
*
* @param path
Expand All @@ -77,15 +56,25 @@ object IOUtils {

/** Reads a file at the given path and:
* - skips BOM if present
* - removes unpaired surrogates
* - uses UTF-8 encoding (replacing malformed and unmappable characters)
*
* @param path
* the file path
* @return
* a String with the given file's contents
*/
def readEntireFile(path: Path): String =
Using.resource(Source.fromFile(path.toFile)(createDecoder()))(contentStringFromBufferedSource)
def readEntireFile(path: Path): String = {
val bytes = Files.readAllBytes(path)
// Strip the UTF-8 BOM (EF BB BF) if present. The String constructor replaces malformed and unmappable input with
// the default replacement character, i.e. decoding is lenient just like in readLinesInFile above.
val offset = if (startsWithUtf8Bom(bytes)) Utf8BomLength else 0
new String(bytes, offset, bytes.length - offset, StandardCharsets.UTF_8)
}

private val Utf8BomLength = 3

private def startsWithUtf8Bom(bytes: Array[Byte]): Boolean =
bytes.length >= Utf8BomLength &&
bytes(0) == 0xef.toByte && bytes(1) == 0xbb.toByte && bytes(2) == 0xbf.toByte

}
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
package io.shiftleft.utils

import org.scalatest.matchers.should.Matchers
import org.scalatest.wordspec.AnyWordSpec

import java.nio.charset.StandardCharsets
import java.nio.file.{Files, Path}

class IOUtilsTest extends AnyWordSpec with Matchers {

private val Utf8Bom = Array(0xef.toByte, 0xbb.toByte, 0xbf.toByte)

private def withTempFile(content: Array[Byte])(f: Path => Unit): Unit = {
val path = Files.createTempFile("io-utils-test", ".tmp")
try {
Files.write(path, content)
f(path)
} finally {
Files.deleteIfExists(path)
}
}

private def utf8(s: String): Array[Byte] = s.getBytes(StandardCharsets.UTF_8)

"IOUtils.readLinesInFile" should {
"read all lines" in {
withTempFile(utf8("foo\nbar\r\nbaz")) { path =>
IOUtils.readLinesInFile(path) shouldBe Seq("foo", "bar", "baz")
}
}

"skip a UTF-8 BOM if present" in {
withTempFile(Utf8Bom ++ utf8("foo\nbar")) { path =>
IOUtils.readLinesInFile(path) shouldBe Seq("foo", "bar")
}
}

"replace malformed UTF-8 input instead of throwing" in {
// 0xC3 starts a two-byte sequence but 0x28 ('(') is not a valid continuation byte
withTempFile(utf8("foo") ++ Array(0xc3.toByte, 0x28.toByte)) { path =>
IOUtils.readLinesInFile(path) shouldBe Seq("foo\ufffd(")
}
}

"not strip a leading character that merely encodes like a BOM" in {
// EF AE BB is the genuine UTF-8 encoding of U+FBBB (a regular character, not a BOM)
withTempFile(Array(0xef.toByte, 0xae.toByte, 0xbb.toByte)) { path =>
IOUtils.readLinesInFile(path) shouldBe Seq("\ufbbb")
}
}

"return an empty Seq for an empty file" in {
withTempFile(Array.emptyByteArray) { path =>
IOUtils.readLinesInFile(path) shouldBe empty
}
}
}

"IOUtils.readEntireFile" should {
"read the entire content" in {
withTempFile(utf8("foo\nbar\n")) { path =>
IOUtils.readEntireFile(path) shouldBe "foo\nbar\n"
}
}

"skip a UTF-8 BOM if present" in {
withTempFile(Utf8Bom ++ utf8("foo")) { path =>
IOUtils.readEntireFile(path) shouldBe "foo"
}
}

"replace malformed UTF-8 input instead of throwing" in {
withTempFile(Array(0xc3.toByte, 0x28.toByte)) { path =>
IOUtils.readEntireFile(path) shouldBe "\ufffd("
}
}

"not strip a leading character that merely encodes like a BOM" in {
// EF AE BB is the genuine UTF-8 encoding of U+FBBB (a regular character, not a BOM)
withTempFile(Array(0xef.toByte, 0xae.toByte, 0xbb.toByte)) { path =>
IOUtils.readEntireFile(path) shouldBe "\ufbbb"
}
}

"return an empty String for an empty file" in {
withTempFile(Array.emptyByteArray) { path =>
IOUtils.readEntireFile(path) shouldBe ""
}
}
}

}
Loading