From 0940fe98cc750a5f50da4023f54db1e826833c53 Mon Sep 17 00:00:00 2001 From: "U-BASIS\\dsmyda" Date: Mon, 20 Jul 2020 10:53:05 -0400 Subject: [PATCH 1/4] Updated the EncodingUtils class to discard Tika's IBM500 result --- .../coreutils/textutils/EncodingUtils.java | 30 +++++++++++++++---- 1 file changed, 24 insertions(+), 6 deletions(-) diff --git a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java index d0bf4a0e36..2562692c57 100755 --- a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java +++ b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java @@ -87,12 +87,30 @@ public class EncodingUtils { try (InputStream stream = new BufferedInputStream(new ReadContentInputStream(file))) { CharsetDetector detector = new CharsetDetector(); detector.setText(stream); - CharsetMatch tikaResult = detector.detect(); - if (tikaResult != null && tikaResult.getConfidence() >= MIN_CHARSETDETECT_MATCH_CONFIDENCE) { - String tikaCharSet = tikaResult.getName(); - //Check if the nio package has support for the charset determined by Tika. - if(Charset.isSupported(tikaCharSet)) { - return Charset.forName(tikaCharSet); + + CharsetMatch[] tikaResults = detector.detectAll(); + // Get all guesses by Tika. These CharsetMatch's are ordered + // by descending confidence (largest first). + if (tikaResults.length > 0) { + // Grab our top pick + CharsetMatch topPick = tikaResults[0]; + + if (topPick.getName().equalsIgnoreCase("IBM500")) { + // Legacy encoding, let's discard this one in favor + // of the second pick. Tika has some problems with + // mistakenly identifying text as IBM500. See JIRA-6600 + // for more details. + if (tikaResults.length > 1) { + topPick = tikaResults[1]; + } + } + + if (!topPick.getName().equalsIgnoreCase("IBM500") && + topPick.getConfidence() >= MIN_CHARSETDETECT_MATCH_CONFIDENCE) { + //Check if the nio package has support for the charset determined by Tika. + if(Charset.isSupported(topPick.getName())) { + return Charset.forName(topPick.getName()); + } } } } From 033935168f7467eea5c7f9fd516f0b06847f5591 Mon Sep 17 00:00:00 2001 From: "U-BASIS\\dsmyda" Date: Mon, 20 Jul 2020 11:33:51 -0400 Subject: [PATCH 2/4] Address Codacy comments --- .../coreutils/textutils/EncodingUtils.java | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java index 2562692c57..e8a918d031 100755 --- a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java +++ b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java @@ -95,22 +95,20 @@ public class EncodingUtils { // Grab our top pick CharsetMatch topPick = tikaResults[0]; - if (topPick.getName().equalsIgnoreCase("IBM500")) { + if (topPick.getName().equalsIgnoreCase("IBM500") && tikaResults.length > 1) { // Legacy encoding, let's discard this one in favor // of the second pick. Tika has some problems with // mistakenly identifying text as IBM500. See JIRA-6600 // for more details. - if (tikaResults.length > 1) { - topPick = tikaResults[1]; - } + topPick = tikaResults[1]; } if (!topPick.getName().equalsIgnoreCase("IBM500") && - topPick.getConfidence() >= MIN_CHARSETDETECT_MATCH_CONFIDENCE) { - //Check if the nio package has support for the charset determined by Tika. - if(Charset.isSupported(topPick.getName())) { - return Charset.forName(topPick.getName()); - } + topPick.getConfidence() >= MIN_CHARSETDETECT_MATCH_CONFIDENCE && + Charset.isSupported(topPick.getName())) { + // Choose this charset since it's supported and has high + // enough confidence + return Charset.forName(topPick.getName()); } } } From 4e7695dc12c75f831d9f13a5af610ab280cc15cf Mon Sep 17 00:00:00 2001 From: "U-BASIS\\dsmyda" Date: Mon, 20 Jul 2020 11:36:37 -0400 Subject: [PATCH 3/4] Removed useless comment --- .../org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java | 1 - 1 file changed, 1 deletion(-) diff --git a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java index e8a918d031..95e9ab20a6 100755 --- a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java +++ b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java @@ -92,7 +92,6 @@ public class EncodingUtils { // Get all guesses by Tika. These CharsetMatch's are ordered // by descending confidence (largest first). if (tikaResults.length > 0) { - // Grab our top pick CharsetMatch topPick = tikaResults[0]; if (topPick.getName().equalsIgnoreCase("IBM500") && tikaResults.length > 1) { From 8532d1dc4c04a4a47648c00d60140b9e65c7fdae Mon Sep 17 00:00:00 2001 From: "U-BASIS\\dsmyda" Date: Mon, 20 Jul 2020 11:43:16 -0400 Subject: [PATCH 4/4] Added link to TIKA story about IBM500 issue --- .../sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java index 95e9ab20a6..e4fc7019c9 100755 --- a/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java +++ b/Core/src/org/sleuthkit/autopsy/coreutils/textutils/EncodingUtils.java @@ -89,7 +89,7 @@ public class EncodingUtils { detector.setText(stream); CharsetMatch[] tikaResults = detector.detectAll(); - // Get all guesses by Tika. These CharsetMatch's are ordered + // Get all guesses by Tika. These matches are ordered // by descending confidence (largest first). if (tikaResults.length > 0) { CharsetMatch topPick = tikaResults[0]; @@ -98,7 +98,8 @@ public class EncodingUtils { // Legacy encoding, let's discard this one in favor // of the second pick. Tika has some problems with // mistakenly identifying text as IBM500. See JIRA-6600 - // for more details. + // and https://issues.apache.org/jira/browse/TIKA-2771 for + // more details. topPick = tikaResults[1]; }