lucene.net + 盘古分词
引用:
1.Lucene.Net.dll
2.PanGu.Lucene.Analyzer.dll
3.PanGu.HighLight.dll
4.PanGu.dll
using Lucene.Net.Search;
using Lucene.Net.Store;
using Lucene.Net.QueryParsers;
using Lucene.Net.Documents;
using Lucene.Net.Index;
using Lucene.Net.Analysis.Standard;
using Lucene.Net.Analysis;
using Lucene.Net.Analysis.PanGu;
using PanGu.HighLight;
using PanGu;
1.建立索引:
static string path = @"G:\indextest";//索引文件储存位置 static void CreateIndex()
{
//创建索引库目录
var directory = FSDirectory.Open(new DirectoryInfo(path));
Analyzer analyzer = null;
//analyzer = new StandardAnalyzer(Lucene.Net.Util.Version.LUCENE_29); if (isPangu)
{
analyzer = new PanGuAnalyzer();//盘古Analyzer
}
else
{
analyzer = new StandardAnalyzer(Lucene.Net.Util.Version.LUCENE_29);
} //创建一个索引,采用StandardAnalyzer对句子进行分词
IndexWriter indexWriter = new IndexWriter(directory, analyzer, true, IndexWriter.MaxFieldLength.LIMITED);
MySqlConnection conn = new MySqlConnection(@"server=localhost;User Id=root;password=123456;Database=ecshop");
conn.Open();
MySqlCommand cmd = new MySqlCommand("select goods_name,goods_brief from ecs_goods", conn);
MySqlDataReader reader = cmd.ExecuteReader();
while (reader.Read())
{
//域的集合:文档,类似于表的行
Document doc = new Document();
//要索引的字段
doc.Add(new Field("goods_name", reader["goods_name"].ToString(), Field.Store.YES, Field.Index.ANALYZED));
doc.Add(new Field("goods_brief", reader["goods_brief"].ToString(), Field.Store.YES, Field.Index.ANALYZED));
indexWriter.AddDocument(doc);
}
reader.Close();
//对索引文件进行优化
indexWriter.Optimize();
indexWriter.Close();
}
2.搜索:
protected void Page_Load(object sender, EventArgs e)
{
keyword = Request.Form["q"];
if (keyword != null && keyword != "")
{
var watch = Stopwatch.StartNew();
Analyzer analyzer = null;
analyzer = new StandardAnalyzer(Lucene.Net.Util.Version.LUCENE_29); //搜索
IndexSearcher searcher = new IndexSearcher(FSDirectory.Open(new DirectoryInfo(path)), true); //查询表达式
QueryParser queryP = new QueryParser(Lucene.Net.Util.Version.LUCENE_29, "goods_name", analyzer); //query.parse:注入查询条件
Query query = queryP.Parse(keyword);
var hits = searcher.Search(query, ); //create highlighter
//IFormatter formatter = new SimpleHTMLFormatter("<span style=\"font-weight:bold;color: red;\">", "</span>");
//SimpleFragmenter fragmenter = new SimpleFragmenter(80);
//var scorer = new QueryScorer(query);
//Highlighter highlighter = new Highlighter(formatter, scorer);
//highlighter.TextFragmenter = fragmenter; //PanGu create highlighter
PanGu.HighLight.SimpleHTMLFormatter simpleHTMLFormatter =
new PanGu.HighLight.SimpleHTMLFormatter("<span style=\"font-weight:bold;color: red;\">", "</span>");
PanGu.HighLight.Highlighter highlighter =
new PanGu.HighLight.Highlighter(simpleHTMLFormatter,
new Segment());
highlighter.FragmentSize = ; for (int i = ; i < hits.totalHits; i++)
{
Document doc = searcher.Doc(hits.scoreDocs[i].doc);
//TokenStream stream = analyzer.TokenStream("goods_name", new StringReader(doc.Get("goods_name")));
//String sample = highlighter.GetBestFragments(stream, doc.Get("goods_name"), 2, "...");
goods g = new goods();
g.goods_name = highlighter.GetBestFragment(keyword, doc.Get("goods_name"));
g.goods_brief = highlighter.GetBestFragment(keyword, doc.Get("goods_brief"));
gs.Add(g);
} watch.Stop(); tasktime = "搜索耗费时间:" + watch.ElapsedMilliseconds + "毫秒";
}
}
多字段搜索
string[] fields = { "Title", "Content" };
MultiFieldQueryParser mq = new MultiFieldQueryParser(Lucene.Net.Util.Version.LUCENE_29, fields, analyzer);
Query multiquery = mq.Parse(keyword);// MultiFieldQueryParser.Parse(Lucene.Net.Util.Version.LUCENE_29, new string[] { keyword }, fields, analyzer);
var hits1 = searcher.Search(multiquery, );