【手把手教你全文检索】Apache Lucene初探

1.jar

<dependency>
<groupId>org.apache.lucene</groupId>
<artifactId>lucene-core</artifactId>
<version>4.0.0</version>
</dependency>
<dependency>
<groupId>org.apache.lucene</groupId>
<artifactId>lucene-analyzers-common</artifactId>
<version>4.0.0</version>
</dependency>
<dependency>
<groupId>org.apache.lucene</groupId>
<artifactId>lucene-highlighter</artifactId>
<version>4.0.0</version>
</dependency>
<dependency>
<groupId>org.apache.lucene</groupId>
<artifactId>lucene-queryparser</artifactId>
<version>4.0.0</version>
</dependency>

2.IndexManager

package jyd.info.controller;

import java.io.BufferedInputStream;
import java.io.BufferedReader;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileReader;
import java.io.InputStreamReader;
import java.io.StringReader;
import java.util.ArrayList;
import java.util.Date;
import java.util.Formatter;
import java.util.List;

import org.apache.lucene.analysis.Analyzer;
import org.apache.lucene.analysis.TokenStream;
import org.apache.lucene.analysis.standard.StandardAnalyzer;
import org.apache.lucene.document.Document;
import org.apache.lucene.document.LongField;
import org.apache.lucene.document.TextField;
import org.apache.lucene.document.Field.Store;
import org.apache.lucene.index.DirectoryReader;
import org.apache.lucene.index.IndexWriter;
import org.apache.lucene.index.IndexWriterConfig;
import org.apache.lucene.queryparser.classic.QueryParser;
import org.apache.lucene.search.IndexSearcher;
import org.apache.lucene.search.Query;
import org.apache.lucene.search.ScoreDoc;
import org.apache.lucene.search.highlight.Fragmenter;
import org.apache.lucene.search.highlight.Highlighter;
import org.apache.lucene.search.highlight.QueryScorer;
import org.apache.lucene.search.highlight.SimpleHTMLFormatter;
import org.apache.lucene.search.highlight.SimpleSpanFragmenter;
import org.apache.lucene.store.Directory;
import org.apache.lucene.store.FSDirectory;
import org.apache.lucene.util.Version;

public class IndexManager {
private static IndexManager indexManager;
private static String content = "";

private static String INDEX_DIR = "D:\\luceneIndex";
private static String DATA_DIR = "D:\\luceneData";
private static Analyzer analyzer = null;
private static Directory directory = null;
private static IndexWriter indexWriter = null;

/**
* 创建索引管理器
*
* @return 返回索引管理器对象
*/
public IndexManager getManager() {
if (indexManager == null) {
this.indexManager = new IndexManager();
}
return indexManager;
}

/**
* 创建当前文件目录的索引
*
* @param path
* 当前文件目录
* @return 是否成功
*/
public static boolean createIndex(String path) {
Date date1 = new Date();
List<File> fileList = getFileList(path);
for (File file : fileList) {
content = "";
// 获取文件后缀
String type = file.getName().substring(
file.getName().lastIndexOf(".") + 1);
if ("txt".equalsIgnoreCase(type)) {

content += txt2String(file);

}
System.out.println("name :" + file.getName());
System.out.println("path :" + file.getPath());
// System.out.println("content :"+content);
System.out.println();
try {
analyzer = new StandardAnalyzer(Version.LUCENE_CURRENT);
directory = FSDirectory.open(new File(INDEX_DIR));
File indexFile = new File(INDEX_DIR);
if (!indexFile.exists()) {
indexFile.mkdirs();
}
IndexWriterConfig config = new IndexWriterConfig(
Version.LUCENE_CURRENT, analyzer);
indexWriter = new IndexWriter(directory, config);

Document document = new Document();
document.add(new TextField("filename", file.getName(),
Store.YES));
document.add(new TextField("content", content, Store.YES));
document.add(new TextField("path", file.getPath(), Store.YES));
indexWriter.addDocument(document);
indexWriter.commit();
closeWriter();

} catch (Exception e) {
e.printStackTrace();
}
content = "";
}
Date date2 = new Date();
System.out.println("创建索引-----耗时：" + (date2.getTime() - date1.getTime())
+ "ms\n");
return true;
}

/**
* 读取txt文件的内容
*
* @param file
* 想要读取的文件对象
* @return 返回文件内容
*/
public static String txt2String(File file) {
String result = "";
try {
String code = codeStringPlus(file.getAbsolutePath());
BufferedReader br = new BufferedReader(new InputStreamReader(new FileInputStream(file), code));// 构造一个BufferedReader类来读取文件
String s = null;
while ((s = br.readLine()) != null) {// 使用readLine方法，一次读一行
result = result + "\n" + s;
}
br.close();
} catch (Exception e) {
e.printStackTrace();
}
return result;
}

/**
* 查找索引，返回符合条件的文件
*
* @param text
* 查找的字符串
* @return 符合条件的文件List
*/
public static void searchIndex(String text) {
Date date1 = new Date();
try {
directory = FSDirectory.open(new File(INDEX_DIR));
analyzer = new StandardAnalyzer(Version.LUCENE_CURRENT);
DirectoryReader ireader = DirectoryReader.open(directory);
IndexSearcher isearcher = new IndexSearcher(ireader);

QueryParser parser = new QueryParser(Version.LUCENE_CURRENT,
"content", analyzer);
Query query = parser.parse(text);
ScoreDoc[] hits = isearcher.search(query, null, 100000000).scoreDocs;

for (int i = 0; i < hits.length; i++) {
Document hitDoc = isearcher.doc(hits[i].doc);
System.out.println(hitDoc.get("filename"));
System.out.println(hitDoc.get("content"));
System.out.println(hitDoc.get("path"));
System.out.println("____________________________");
}
ireader.close();
} catch (Exception e) {
e.printStackTrace();
}
Date date2 = new Date();
System.out.println("查看索引-----耗时：" + (date2.getTime() - date1.getTime())
+ "ms\n");
}

/**
* 过滤目录下的文件
*
* @param dirPath
* 想要获取文件的目录
* @return 返回文件list
*/
public static List<File> getFileList(String dirPath) {
File[] files = new File(dirPath).listFiles();
List<File> fileList = new ArrayList<File>();
for (File file : files) {
if (isTxtFile(file.getName())) {
fileList.add(file);
}
}
return fileList;
}

/**
* 判断是否为目标文件，目前支持txt xls doc格式
*
* @param fileName
* 文件名称
* @return 如果是文件类型满足过滤条件，返回true；否则返回false
*/
public static boolean isTxtFile(String fileName) {
if (fileName.lastIndexOf(".txt") > 0) {
return true;
} else if (fileName.lastIndexOf(".xls") > 0) {
return true;
} else if (fileName.lastIndexOf(".doc") > 0) {
return true;
}
return false;
}

public static void closeWriter() throws Exception {
if (indexWriter != null) {
indexWriter.close();
}
}

/**
* 删除文件目录下的所有文件
*
* @param file
* 要删除的文件目录
* @return 如果成功，返回true.
*/
public static boolean deleteDir(File file) {
if (file.isDirectory()) {
File[] files = file.listFiles();
for (int i = 0; i < files.length; i++) {
deleteDir(files[i]);
}
}
file.delete();
return true;
}

/**
* 查询字符编码
* @param fileName
* @return UTF-8/Unicode/UTF-16BE/GBK
* @throws Exception
*/
public static String codeStringPlus(String fileName) throws Exception {
BufferedInputStream bin = null;
String code = null;

try {
bin = new BufferedInputStream(new FileInputStream(fileName));
int p = (bin.read() << 8) + bin.read();
switch (p) {
case 0xefbb:
code = "UTF-8";
break;
case 0xfffe:
code = "Unicode";
break;
case 0xfeff:
code = "UTF-16BE";
break;
default:
code = "GBK";
}
} catch (Exception e) {
e.printStackTrace();
} finally {
bin.close();
}

return code;
}

public static void main(String[] args) {
File fileIndex = new File(INDEX_DIR);
if (deleteDir(fileIndex)) {
fileIndex.mkdir();
} else {
fileIndex.mkdir();
}

createIndex(DATA_DIR);
searchIndex("涂料");
}

}

3.结果：

注释：可以改进，读取excell及doc的内容，懒得改，以后优化。

文章来源：http://www.cnblogs.com/xing901022/p/3933675.html

posted @ 2016-12-23 16:23 wangkejun 阅读(520) 评论(0) 收藏举报

刷新页面返回顶部

wangkejun

【手把手教你全文检索】Apache Lucene初探

公告