Merge pull request #2735 from millmanorama/2543-use-kws-type-in-indexedtext-paging

2543 use kws type in indexedtext paging
This commit is contained in:
Richard Cordovano
2017-04-26 12:44:33 -04:00
committed by GitHub
2 changed files with 92 additions and 54 deletions
@@ -18,23 +18,19 @@
*/
package org.sleuthkit.autopsy.keywordsearch;
import com.google.common.base.Predicate;
import com.google.common.collect.Iterators;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Collection;
import java.util.HashMap;
import java.util.HashSet;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Optional;
import java.util.Set;
import java.util.TreeMap;
import java.util.TreeSet;
import java.util.logging.Level;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import javax.annotation.concurrent.GuardedBy;
import org.apache.commons.lang.StringUtils;
import org.apache.commons.lang3.StringUtils;
import org.apache.solr.client.solrj.SolrQuery;
import org.apache.solr.client.solrj.SolrRequest.METHOD;
import org.apache.solr.client.solrj.response.QueryResponse;
@@ -89,7 +85,8 @@ class AccountsText implements IndexedText {
*/
private final TreeMap<Integer, Integer> numberOfHitsPerPage = new TreeMap<>();
/*
* set of pages, used for iterating back and forth. Only stores pages with hits
* set of pages, used for iterating back and forth. Only stores pages with
* hits
*/
private final Set<Integer> pages = numberOfHitsPerPage.keySet();
/*
@@ -142,7 +139,7 @@ class AccountsText implements IndexedText {
@NbBundle.Messages("AccountsText.nextPage.exception.msg=No next page.")
public int nextPage() {
if (hasNextPage()) {
currentPage =Iterators.get(pages.iterator(),getIndexOfCurrentPage() + 1);
currentPage = Iterators.get(pages.iterator(), getIndexOfCurrentPage() + 1);
return currentPage;
} else {
throw new IllegalStateException(Bundle.AccountsText_nextPage_exception_msg());
@@ -153,7 +150,7 @@ class AccountsText implements IndexedText {
@NbBundle.Messages("AccountsText.previousPage.exception.msg=No previous page.")
public int previousPage() {
if (hasPreviousPage()) {
currentPage = Iterators.get(pages.iterator(),getIndexOfCurrentPage() - 1);
currentPage = Iterators.get(pages.iterator(), getIndexOfCurrentPage() - 1);
return currentPage;
} else {
throw new IllegalStateException(Bundle.AccountsText_previousPage_exception_msg());
@@ -215,52 +212,72 @@ class AccountsText implements IndexedText {
* Initialize this object with information about which pages/chunks have
* hits. Multiple calls will not change the initial results.
*/
synchronized private void loadPageInfo() throws IllegalStateException, TskCoreException {
synchronized private void loadPageInfo() throws IllegalStateException, TskCoreException, KeywordSearchModuleException, NoOpenCoreException {
if (isPageInfoLoaded) {
return;
}
try {
this.numberPagesForFile = solrServer.queryNumFileChunks(this.solrObjectId);
} catch (KeywordSearchModuleException | NoOpenCoreException ex) {
logger.log(Level.WARNING, "Could not get number pages for content " + this.solrObjectId, ex); //NON-NLS
return;
}
this.numberPagesForFile = solrServer.queryNumFileChunks(this.solrObjectId);
boolean needsQuery = false;
for (BlackboardArtifact artifact : artifacts) {
addToPagingInfo(artifact);
if (solrObjectId != artifact.getObjectID()) {
throw new IllegalStateException("not all artifacts are from the same object!");
}
//add both the canonical form and the form in the text as accountNumbers to highlight.
this.accountNumbers.add(artifact.getAttribute(TSK_KEYWORD).getValueString());
this.accountNumbers.add(artifact.getAttribute(TSK_CARD_NUMBER).getValueString());
//if the chunk id is present just use that.
Optional<Integer> chunkID =
Optional.ofNullable(artifact.getAttribute(TSK_KEYWORD_SEARCH_DOCUMENT_ID))
.map(BlackboardAttribute::getValueString)
.map(String::trim)
.map(kwsdocID -> StringUtils.substringAfterLast(kwsdocID, Server.CHUNK_ID_SEPARATOR))
.map(Integer::valueOf);
if (chunkID.isPresent()) {
numberOfHitsPerPage.put(chunkID.get(), 0);
currentHitPerPage.put(chunkID.get(), 0);
} else {
//otherwise we need to do a query to figure out the paging.
needsQuery = true;
}
}
if (needsQuery) {
// Run a query to figure out which chunks for the current object have hits.
Keyword queryKeyword = new Keyword(CCN_REGEX, false, false);
KeywordSearchQuery chunksQuery = KeywordSearchUtil.getQueryForKeyword(queryKeyword, new KeywordList(Arrays.asList(queryKeyword)));
chunksQuery.addFilter(new KeywordQueryFilter(KeywordQueryFilter.FilterType.CHUNK, this.solrObjectId));
//load the chunks/pages from the result of the query.
loadPageInfoFromHits(chunksQuery.performQuery());
}
this.currentPage = pages.stream().findFirst().orElse(1);
isPageInfoLoaded = true;
}
private static final String CCN_REGEX = "(%?)(B?)([0-9][ \\-]*?){12,19}(\\^?)";
private void addToPagingInfo(BlackboardArtifact artifact) throws IllegalStateException, TskCoreException {
if (solrObjectId != artifact.getObjectID()) {
throw new IllegalStateException("not all artifacts are from the same object!");
/**
* Load the paging info from the QueryResults object.
*/
synchronized private void loadPageInfoFromHits(QueryResults hits) {
//organize the hits by page, filter as needed
for (Keyword k : hits.getKeywords()) {
for (KeywordHit hit : hits.getResults(k)) {
int chunkID = hit.getChunkId();
if (chunkID != 0 && this.solrObjectId == hit.getSolrObjectId()) {
String hitString = hit.getHit();
if (accountNumbers.stream().anyMatch(hitString::contains)) {
numberOfHitsPerPage.put(chunkID, 0); //unknown number of matches in the page
currentHitPerPage.put(chunkID, 0); //set current hit to 0th
}
}
}
}
accountNumbers.add(artifact.getAttribute(TSK_CARD_NUMBER).getValueString());
final BlackboardAttribute keywordAttribute = artifact.getAttribute(TSK_KEYWORD);
if (keywordAttribute != null) {
accountNumbers.add(keywordAttribute.getValueString());
}
List<String> rawDocIDs = new ArrayList<>();
final BlackboardAttribute docID = artifact.getAttribute(TSK_KEYWORD_SEARCH_DOCUMENT_ID);
if (docID != null) {
rawDocIDs.add(docID.getValueString());
}
rawDocIDs.stream()
.map(String::trim)
.map(t -> StringUtils.substringAfterLast(t, Server.CHUNK_ID_SEPARATOR))
.map(Integer::valueOf)
.forEach(chunkID -> {
numberOfHitsPerPage.put(chunkID, 0);
currentHitPerPage.put(chunkID, 0);
});
}
@Override
@@ -288,8 +305,8 @@ class AccountsText implements IndexedText {
QueryResponse queryResponse = solrServer.query(q, METHOD.POST);
String highlightedText
= HighlightedText.attemptManualHighlighting(
String highlightedText =
HighlightedText.attemptManualHighlighting(
queryResponse.getResults(),
Server.Schema.CONTENT_STR.toString(),
accountNumbers
@@ -41,7 +41,6 @@ import org.openide.util.NbBundle.Messages;
import org.sleuthkit.autopsy.coreutils.Logger;
import org.sleuthkit.autopsy.coreutils.Version;
import org.sleuthkit.autopsy.keywordsearch.KeywordQueryFilter.FilterType;
import org.sleuthkit.autopsy.keywordsearch.KeywordSearch.QueryType;
import org.sleuthkit.datamodel.BlackboardArtifact;
import org.sleuthkit.datamodel.BlackboardAttribute;
import org.sleuthkit.datamodel.TskCoreException;
@@ -59,6 +58,7 @@ class HighlightedText implements IndexedText {
private static final BlackboardAttribute.Type TSK_KEYWORD_SEARCH_TYPE = new BlackboardAttribute.Type(BlackboardAttribute.ATTRIBUTE_TYPE.TSK_KEYWORD_SEARCH_TYPE);
private static final BlackboardAttribute.Type TSK_KEYWORD = new BlackboardAttribute.Type(BlackboardAttribute.ATTRIBUTE_TYPE.TSK_KEYWORD);
static private final BlackboardAttribute.Type TSK_ASSOCIATED_ARTIFACT = new BlackboardAttribute.Type(BlackboardAttribute.ATTRIBUTE_TYPE.TSK_ASSOCIATED_ARTIFACT);
static private final BlackboardAttribute.Type TSK_KEYWORD_REGEXP = new BlackboardAttribute.Type(BlackboardAttribute.ATTRIBUTE_TYPE.TSK_KEYWORD_REGEXP);
private static final String HIGHLIGHT_PRE = "<span style='background:yellow'>"; //NON-NLS
private static final String HIGHLIGHT_POST = "</span>"; //NON-NLS
@@ -175,13 +175,21 @@ class HighlightedText implements IndexedText {
qt = (queryTypeAttribute != null)
? KeywordSearch.QueryType.values()[queryTypeAttribute.getValueInt()] : null;
isLiteral = qt != QueryType.REGEX;
Keyword keywordQuery = null;
switch (qt) {
case LITERAL:
case SUBSTRING:
keywordQuery = new Keyword(keyword, true, true);
break;
case REGEX:
String regexp = artifact.getAttribute(TSK_KEYWORD_REGEXP).getValueString();
keywordQuery = new Keyword(regexp, false, false);
break;
}
KeywordSearchQuery chunksQuery = KeywordSearchUtil.getQueryForKeyword(keywordQuery, new KeywordList(Arrays.asList(keywordQuery)));
// Run a query to figure out which chunks for the current object have
// hits for this keyword.
Keyword keywordQuery = new Keyword(keyword, isLiteral, true);
KeywordSearchQuery chunksQuery = new LuceneQuery(new KeywordList(Arrays.asList(keywordQuery)), keywordQuery);
chunksQuery.escape();
chunksQuery.addFilter(new KeywordQueryFilter(FilterType.CHUNK, this.objectId));
hits = chunksQuery.performQuery();
@@ -197,11 +205,24 @@ class HighlightedText implements IndexedText {
for (Keyword k : hits.getKeywords()) {
for (KeywordHit hit : hits.getResults(k)) {
int chunkID = hit.getChunkId();
if (chunkID != 0 && this.objectId == hit.getSolrObjectId()) {
numberOfHitsPerPage.put(chunkID, 0); //unknown number of matches in the page
currentHitPerPage.put(chunkID, 0); //set current hit to 0th
if (StringUtils.isNotBlank(hit.getHit())) {
this.keywords.add(hit.getHit());
if (artifact != null) {
if (chunkID != 0 && this.objectId == hit.getSolrObjectId()) {
String hit1 = hit.getHit();
if (keywords.stream().anyMatch(hit1::contains)) {
numberOfHitsPerPage.put(chunkID, 0); //unknown number of matches in the page
currentHitPerPage.put(chunkID, 0); //set current hit to 0th
}
}
} else {
if (chunkID != 0 && this.objectId == hit.getSolrObjectId()) {
numberOfHitsPerPage.put(chunkID, 0); //unknown number of matches in the page
currentHitPerPage.put(chunkID, 0); //set current hit to 0th
if (StringUtils.isNotBlank(hit.getHit())) {
this.keywords.add(hit.getHit());
}
}
}
}