Lucene.Net 与 盘古分词
2021-06-29 21:05
指定Field.Store.YES的字段在检索时才干用document.Get取出原值
//Field.Index.NOT_ANALYZED:指定不依照分词后的结果保存--是否按分词后结果保存取决于是否对该列内容进行模糊查询
document.Add(new Field("TextZh", dt.Rows[n]["TextZh"].ToString(), Field.Store.YES, Field.Index.ANALYZED, Field.TermVector.WITH_POSITIONS_OFFSETS));
//Field.Index.ANALYZED:指定文章内容依照分词后结果保存 否则无法实现兴许的模糊查询
//WITH_POSITIONS_OFFSETS:指示不仅保存切割后的词 还保存词之间的距离
//document.Add(new Field("content", "我常常出去玩", Field.Store.YES, Field.Index.ANALYZED, Field.TermVector.WITH_POSITIONS_OFFSETS));
writer.AddDocument(document); //文档写入索引库
Console.Write("{0}\r", n+1);
}
}
writer.Close();//会自己主动解锁
directory.Close(); //不要忘了Close,否则索引结果搜不到
}
public static void Query(string strQuery)
{
Stopwatch sw = new Stopwatch();
sw.Start();
FSDirectory directory = FSDirectory.Open(new DirectoryInfo("CH-EG"), new NoLockFactory());
IndexReader reader = IndexReader.Open(directory, true);
IndexSearcher searcher = new IndexSearcher(reader);
//搜索条件
PhraseQuery query = new PhraseQuery();
//把用户输入的关键字进行分词
foreach(string word in SplitWords(strQuery)) {
query.Add(new Term("TextZh", word));
}
//query.Add(new Term("content", "C#"));//多个查询条件时 为且的关系
query.SetSlop(100); //指定关键词相隔最大距离
//TopScoreDocCollector盛放查询结果的容器
TopScoreDocCollector collector = TopScoreDocCollector.create(1000, true);
searcher.Search(query, null, collector);//依据query查询条件进行查询,查询结果放入collector容器
sw.Stop();
//TopDocs 指定0到GetTotalHits() 即全部查询结果中的文档 假设TopDocs(20,10)则意味着获取第20-30之间文档内容 达到分页的效果
ScoreDoc[] docs = collector.TopDocs(0, collector.GetTotalHits()).scoreDocs;
//展示数据实体对象集合
for (int i = 0; i
{
int docId = docs[i].doc;//得到查询结果文档的id(Lucene内部分配的id)
Document doc = searcher.Doc(docId);//依据文档id来获得文档对象Document
Console.Write("{0}\n", doc.Get("TextZh"));
}
TimeSpan ts2 = sw.Elapsed;
Console.WriteLine("本次查询总共花费{0}ms.\n", ts2.TotalMilliseconds);
}
static void Main(string[] args)
{
//CreateIndex("CH-EG");
Console.Write("Press phrase: \n");
string strQuery = Console.ReadLine();
while (strQuery != "")
{
Query(strQuery);
Console.Write("Press phrase: \n");
strQuery = Console.ReadLine();
}
}
}
}
using System;
using System.IO;
using System.Collections.Generic;
using System.Linq;
using System.Text;
using Lucene.Net.Analysis;
using Lucene.Net.Documents;
using Lucene.Net.Index;
using Lucene.Net.Search;
using Lucene.Net.Store;
using Lucene.Net.Analysis.PanGu;
using Maticsoft.DBUtility;
using System.Data;
using System.Diagnostics;
namespace SearchSentence
{
class Program
{
public static string[] SplitWords(string content)
{
List
上一篇:css float浮动清除
下一篇:HTML中的单位小结