对lucene应用的总结
Lucene 是一个基于 Java 的全文信息检索工具包,它不是一个完整的搜索应用程序,而是为你的应用程序提供索引和搜索功能。Lucene 目前是 Apache Jakarta 家族中的一个开源项目。也是目前最为流行的基于 Java 开源全文检索工具包。
功能:将blog文章进行检索管理,该实例注重参考lucene函数使用和索引文件安全维护!
<dependency><groupId>lucene</groupId>
<artifactId>lucene-core</artifactId>
<version>2.0.0</version>
</dependency>
<dependency>
<groupId>open-source.missing.org.apache.lucene</groupId>
<artifactId>lucene-highlighter</artifactId>
<version>2.3.2</version>
</dependency>
<dependency>
<groupId>open-source.missing.org.mira.lucene</groupId>
<artifactId>IKAnalyzer</artifactId>
<version>2.0.2OBF</version>
</dependency>
package com.infowarelab.best1.util.lucene.blog;
import java.io.File;
import java.io.IOException;
import java.io.Reader;
import java.io.StringReader;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import org.apache.commons.logging.Log;
import org.apache.commons.logging.LogFactory;
import org.apache.lucene.analysis.Analyzer;
import org.apache.lucene.analysis.Token;
import org.apache.lucene.analysis.TokenStream;
import org.apache.lucene.document.Document;
import org.apache.lucene.document.Field;
import org.apache.lucene.index.IndexReader;
import org.apache.lucene.index.IndexWriter;
import org.apache.lucene.index.Term;
import org.apache.lucene.queryParser.ParseException;
import org.apache.lucene.queryParser.QueryParser;
import org.apache.lucene.search.BooleanClause;
import org.apache.lucene.search.BooleanQuery;
import org.apache.lucene.search.Hits;
import org.apache.lucene.search.IndexSearcher;
import org.apache.lucene.search.Query;
import org.apache.lucene.search.Sort;
import org.apache.lucene.search.SortField;
import org.apache.lucene.store.Directory;
import org.apache.lucene.store.FSDirectory;
import org.mira.lucene.analysis.MIK_CAnalyzer;
import org.springside.core.dao.support.Page;
import com.infowarelab.best1.exception.BusinessException;
import com.infowarelab.best1.persistence.dao.blog.BlogEntryDao;
import com.infowarelab.best1.persistence.domain.blog.BlogEntry;
import com.infowarelab.best1.web.listener.DroolsSpringBeanManager;

/** *//**
* 功能描述: 对博客文章进行全文检索管理
* @author :weijie.yang 2008-06-16
*/
public class LuceneManager
{
private static LuceneManager instance ;
private static final Log logger = LogFactory.getLog(LuceneManager.class);
private static final String OPER_WRITE="write";
private static final String OPER_REMOVE="remove";
private static final String FIELD_ID = "id";
private static final String FIELD_TITLE = "title";
private static final String FIELD_TEXT = "summary";

private LuceneManager()
{}

synchronized public static LuceneManager getInstance()
{
if(instance == null)
instance = new LuceneManager();
return instance;
}
/** *//**
*
* 功能描述:获取当前用户索引目录
* @param indexFoldPath索引文件保存路径
* @return
*/
private File getIndexFolder(String indexFoldPath)
{
File file = new File(indexFoldPath);
if (!file.exists())
{
file.mkdirs();
}
return file;
}

/** *//**
* 功能描述:获取IndexWriter
* @param indexFoldPath索引文件保存路径
* @return
*/
private IndexWriter getIndexWriter(String indexFoldPath)
{
IndexWriter writer = null;
try
{
Analyzer analyzer = new MIK_CAnalyzer();
File indexFolder = getIndexFolder(indexFoldPath);
File file = new File(indexFolder, "segments");
if (file.exists())
{
writer = new IndexWriter(indexFolder, analyzer, false);
} else
{
writer = new IndexWriter(indexFolder, analyzer, true);
} 
} catch (Exception e)
{
logger.error("get IndexWriter error("+indexFoldPath+"):"+e.getStackTrace());
}
return writer;
}

/** *//**
*
* 功能描述:建立索引
* @param blogEntrys
* @param indexFoldPath
* @throws Exception
*/
private void writeToIndex(List<BlogEntry> blogEntrys,String indexFoldPath) throws Exception
{
IndexWriter writer = getIndexWriter(indexFoldPath);
try
{
for (BlogEntry blogEntry : blogEntrys)
{
Document document = new Document();
document.add(new Field(FIELD_ID, String.valueOf(blogEntry
.getId()), Field.Store.YES, Field.Index.UN_TOKENIZED));
document.add(new Field("siteId", String.valueOf(blogEntry
.getSite().getId()), Field.Store.YES, Field.Index.UN_TOKENIZED));
document.add(new Field(FIELD_TITLE, String.valueOf(blogEntry
.getTitle()), Field.Store.YES, Field.Index.TOKENIZED));
document.add(new Field(FIELD_TEXT, String.valueOf(blogEntry
.getText()), Field.Store.YES, Field.Index.TOKENIZED));
writer.addDocument(document);
}
writer.optimize();
} catch (Exception e)
{
logger.error("error on build index", e);
throw new BusinessException("error on build index", e);
} finally
{
if (writer != null)
{
try
{
writer.close();
} catch (IOException e)
{
logger
.error(
"writeToIndex(List blogEntrys) - unable to close writer",
e);
}
}
}
}

/** *//**
*
* 功能描述:删除索引
* @param blogEntryId
* @param siteId
*/
private void removeFromIndex(Long blogEntryId,String indexFoldPath)
{
IndexReader reader = null;
Directory directory = null;
try
{
Term term = new Term(FIELD_ID, String.valueOf(blogEntryId));
File indexFolder = getIndexFolder(indexFoldPath);
File file = new File(indexFolder, "segments");

if (file.exists())
{
directory = FSDirectory.getDirectory(indexFolder, false);
} else
{
directory = FSDirectory.getDirectory(indexFolder, true);
}
reader = IndexReader.open(directory);
reader.deleteDocuments(term);
logger.info("remove index success");
} catch (IOException e)
{
logger.warn(
"remove(Long) - error at remove blogEntry by blogEntryId - blogEntryId="
+ blogEntryId, e);
} finally
{
if (reader != null)
{
try
{
reader.close();
} catch (IOException e)
{
logger.error("remove(Long) - unable to close reader", e);
}
}
}
}

/** *//**
* 功能描述:由于lucene并发控制不允许indexWriter和indexReader同时对索引文件进行修改.
* 对writeToIndex和removeFromIndex方法进行同步处理
* @param flag :添加,删除的标志
* @param obj
* @throws Exception
*/
private synchronized void writeOrRemove(String flag, Object obj,String indexFoldPath)
{
if (flag != null && flag.equals(OPER_WRITE))
{
if (obj != null && obj instanceof List)
{
List<BlogEntry> blogEntrys = (List<BlogEntry>) obj;
try
{
writeToIndex(blogEntrys,indexFoldPath);
} catch (Exception e)
{
logger.error("error on writeToIndex:", e);
}
}
}
if (flag != null && flag.equals(OPER_REMOVE))
{
if (obj != null && obj instanceof Long)
{
Long blogEntryId = (Long) obj;
removeFromIndex(blogEntryId,indexFoldPath);
}
}
}

public void add(BlogEntry blogEntry,String indexFoldPath)
{
try
{
writeOrRemove(OPER_WRITE,Arrays.asList(new BlogEntry[]
{ blogEntry }),indexFoldPath);
logger.info("add index success");
} catch (Exception e)
{
e.printStackTrace();
}
}

public void remove(BlogEntry blogEntry,String indexFoldPath)
{
writeOrRemove(OPER_REMOVE,blogEntry.getId(),indexFoldPath);
}

public void remove(List<BlogEntry> blogEntrys,String indexFoldPath)
{
for (BlogEntry blogEntry : blogEntrys)
{
remove(blogEntry,indexFoldPath);
}
}

public void remove(String ids,String indexFoldPath)
{
String[] idList = ids.split(",");
for (int i = 0; i < idList.length; i++)
{
String id = idList[i];
writeOrRemove(OPER_REMOVE,Long.valueOf(id),indexFoldPath);
}
}

public void update(BlogEntry blogEntry,String indexFoldPath)
{
if (blogEntry == null)
{
return;
}
try
{
remove(blogEntry,indexFoldPath);
add(blogEntry,indexFoldPath);
logger.info("update index success");
} catch (Exception e)
{
logger.warn("update(Long) - error at update blogEntry ");
}
}

/** *//**
* 功能描述:删除目录下所有文件
* @param path
*/
public static void deleteAll(File path)
{
if (!path.exists())
return;
if (path.isFile())
{
path.delete();
return;
}
File[] files = path.listFiles();
for (int i = 0; i < files.length; i++)
{
deleteAll(files[i]);
}
}

/** *//**
*
* 功能描述:对外搜索方法
* @param keyword
* @param pageNo
* @param pageSize
* @param indexFoldPath
* @return
*/
public Page search(String keyword, int pageNo,
int pageSize,String indexFoldPath)
{
Page page = null;
try
{
page = getResultPage(keyword, pageNo, pageSize,indexFoldPath);
} catch (Exception e)
{
e.printStackTrace();
}
return page;
}
private Page getResultPage(String keyword, int pageNo,
int pageSize,String indexFoldPath)
{
IndexSearcher searcher = null;
Hits hits = null;
try
{
Query query = getQuery(keyword);
searcher = getIndexSearcher(indexFoldPath);
//按id倒序排列,lucene默认是按照匹配度排序的.
hits = searcher.search(query,new Sort(new SortField(FIELD_ID,true)));
int startIndex = (pageNo - 1) * pageSize;
int endIndex = pageNo * pageSize - 1;
int totalCount = hits.length();
List<BlogEntry> docs = new ArrayList<BlogEntry>();
if (endIndex >= hits.length())
endIndex = hits.length() - 1;

for (int i = startIndex; i <= endIndex; i++)
{
Document doc = hits.doc(i);
String id = doc.get(FIELD_ID);
BlogEntryDao blogEntryDao = (BlogEntryDao) DroolsSpringBeanManager
.getBean("blogEntryDao");
BlogEntry blogEntry = (BlogEntry) blogEntryDao.get(Long
.valueOf(id));
docs.add(blogEntry);
}
Page page = new Page(startIndex, totalCount, pageSize, docs);
return page;
} catch (Exception e)
{
e.printStackTrace();
} finally
{
if (searcher != null)
{
try
{
searcher.close();
} catch (IOException e)
{
e.printStackTrace();
}
}
}
return null;
}

private Query getQuery(String keyword)
{
BooleanQuery query = new BooleanQuery();
addQuery(query, keyword, FIELD_TITLE);
addQuery(query, keyword, FIELD_TEXT);
return query;
}

private void addQuery(BooleanQuery query, String keyword, String field)
{
QueryParser parse = new QueryParser(field, new MIK_CAnalyzer());
try
{
// 提高搜索精确度
query.add(parse.parse(LuceneUtil.dealKeywords(keyword)),
BooleanClause.Occur.SHOULD);
} catch (ParseException e)
{
if (logger.isDebugEnabled())
{
logger
.debug("getQuery(BooleanQuery, String, String) - parse error - keyword="
+ keyword + ", field=" + field);
}
}
}

private IndexSearcher getIndexSearcher(String indexFoldPath)
{
try
{
IndexSearcher searcher = new IndexSearcher(getIndexFolder(indexFoldPath).getPath());
return searcher;
} catch (IOException ex)
{
logger.error("getIndexSearcher() - error to get searchIndexer", ex);
throw new BusinessException("error to get searchIndexer", ex);
}
}

/** *//**
* 将搜索输入的文字分词,分为多个关键字
*
* @param text
* @return
*/
@SuppressWarnings("unchecked")
public static List getKeywords(String text)
{
List list = new ArrayList();
Analyzer analyzer = new MIK_CAnalyzer();
Reader r = new StringReader(text);
TokenStream ts = (TokenStream) analyzer.tokenStream("", r);
Token t;
try
{
while ((t = ts.next()) != null)
{
list.add(t.termText());
}
} catch (IOException e)
{
e.printStackTrace();
}
return list;
}

/** *//**
* 将搜索输入的文字分词,并将分隔的词用空格连接,这样搜索的结果将更多些
*
* @param text
* @return
*/
public static String dealKeywords(String text)
{
String result = "";
Analyzer analyzer = new MIK_CAnalyzer();
Reader r = new StringReader(text);
TokenStream ts = (TokenStream) analyzer.tokenStream("", r);
Token t;
try
{
while ((t = ts.next()) != null)
{
result += t.termText() + " ";
}
result = result.substring(0, result.length() - 1);
} catch (IOException e)
{
result = "";
e.printStackTrace();
}
return result;
}
}

浙公网安备 33010602011771号