Lucene.Net (4.8) 自动完成/自动建议

Lucene.Net (4.8) AutoComplete / AutoSuggestion

我想使用 Lucene.Net 4.8 实现可搜索索引,为用户提供单个单词和短语的建议/自动完成。

索引创建成功;这些建议是我停滞不前的地方。

4.8 版似乎引入了大量重大更改,并且 none 我找到的可用示例有效。

我的立场

供参考,LuceneVersion是这样的:

private readonly LuceneVersion LuceneVersion = LuceneVersion.LUCENE_48;

解决方案 1

I've tried this,却过不了reader.Terms:

    public void TryAutoComplete()
    {
        var analyzer = new EnglishAnalyzer(LuceneVersion);
        var config = new IndexWriterConfig(LuceneVersion, analyzer);
        RAMDirectory dir = new RAMDirectory();
        using (IndexWriter iw = new IndexWriter(dir, config))
        {
            Document d = new Document();
            TextField f = new TextField("text","",Field.Store.YES);
            d.Add(f);
            f.SetStringValue("abc");
            iw.AddDocument(d);
            f.SetStringValue("colorado");
            iw.AddDocument(d);
            f.SetStringValue("coloring book");
            iw.AddDocument(d);
            iw.Commit();
            using (IndexReader reader = iw.GetReader(false))
            {
                TermEnum terms = reader.Terms(new Term("text", "co"));
                int maxSuggestsCpt = 0;
                // will print:
                // colorado
                // coloring book
                do
                {
                    Console.WriteLine(terms.Term.Text);
                    maxSuggestsCpt++;
                    if (maxSuggestsCpt >= 5)
                        break;
                }
                while (terms.Next() && terms.Term.Text.StartsWith("co"));
            }
        }
    }

reader.Termsno longer exists。作为 Lucene 的新手,不清楚如何重构它。

解决方案 2

正在尝试 this,但出现错误:

    public void TryAutoComplete2()
    {
        using(var analyzer = new EnglishAnalyzer(LuceneVersion))
        {
            IndexWriterConfig config = new IndexWriterConfig(LuceneVersion, analyzer);
            RAMDirectory dir = new RAMDirectory();
            using(var iw = new IndexWriter(dir,config))
            {
                Document d = new Document()
                {
                    new TextField("text", "this is a document with a some words",Field.Store.YES),
                    new Int32Field("id", 42, Field.Store.YES)
                };

                iw.AddDocument(d);
                iw.Commit();

                using (IndexReader reader = iw.GetReader(false))
                using (SpellChecker speller = new SpellChecker(new RAMDirectory()))
                {
                    //ERROR HERE!!!
                    speller.IndexDictionary(new LuceneDictionary(reader, "text"), config, false);
                    string[] suggestions = speller.SuggestSimilar("dcument", 5);
                    IndexSearcher searcher = new IndexSearcher(reader);
                    foreach (string suggestion in suggestions)
                    {
                        TopDocs docs = searcher.Search(new TermQuery(new Term("text", suggestion)), null, Int32.MaxValue);
                        foreach (var doc in docs.ScoreDocs)
                        {
                            System.Diagnostics.Debug.WriteLine(searcher.Doc(doc.Doc).Get("id"));
                        }
                    }
                }
            }
        }
    }

调试时,speller.IndexDictionary(new LuceneDictionary(reader, "text"), config, false); 抛出 The object cannot be set twice! 错误,我无法解释。

欢迎任何想法。

澄清

我想要 return 给定输入的建议术语列表,而不是文档或其全部内容。

例如,如果文档包含 "Hello, my name is Clark. I'm from Atlanta,",我提交了 "Atl,",那么 "Atlanta" 应该作为建议返回。

如果我对您的理解正确的话,您的索引设计可能有点过于复杂了。如果您的目标是将 Lucene 用于 auto-complete,您希望为您认为 complete 的术语创建一个索引。然后使用部分单词或短语使用 PrefixQuery 简单地查询索引。

using Lucene.Net.Analysis;
using Lucene.Net.Analysis.En;
using Lucene.Net.Documents;
using Lucene.Net.Index;
using Lucene.Net.Search;
using Lucene.Net.Store;
using Lucene.Net.Util;
using System;
using System.Linq;

namespace LuceneDemoApp
{
    class LuceneAutoCompleteIndex : IDisposable
    {
        const LuceneVersion Version = LuceneVersion.LUCENE_48;
        RAMDirectory Directory;
        Analyzer Analyzer;
        IndexWriterConfig WriterConfig;

        private void IndexDoc(IndexWriter writer, string term)
        {
            Document doc = new Document();
            doc.Add(new StringField(FieldName, term, Field.Store.YES));
            writer.AddDocument(doc);
        }

        public LuceneAutoCompleteIndex(string fieldName, int maxResults)
        {
            FieldName = fieldName;
            MaxResults = maxResults;
            Directory = new RAMDirectory();
            Analyzer = new EnglishAnalyzer(Version);
            WriterConfig = new IndexWriterConfig(Version, Analyzer);
            WriterConfig.OpenMode = OpenMode.CREATE_OR_APPEND;
        }

        public string FieldName { get; }
        public int MaxResults { get; set; }

        public void Add(string term)
        {
            using (var writer = new IndexWriter(Directory, WriterConfig))
            {
                IndexDoc(writer, term);
            }
        }

        public void AddRange(string[] terms)
        {
            using (var writer = new IndexWriter(Directory, WriterConfig))
            {
                foreach (string term in terms)
                {
                    IndexDoc(writer, term);
                }
            }
        }

        public string[] WhereStartsWith(string term)
        {
            using (var reader = DirectoryReader.Open(Directory))
            {
                IndexSearcher searcher = new IndexSearcher(reader);
                var query = new PrefixQuery(new Term(FieldName, term));
                TopDocs foundDocs = searcher.Search(query, MaxResults);
                var matches = foundDocs.ScoreDocs
                    .Select(scoreDoc => searcher.Doc(scoreDoc.Doc).Get(FieldName))
                    .ToArray();

                return matches;
            }
        }

        public void Dispose()
        {
            Directory.Dispose();
            Analyzer.Dispose();
        }
    }
}

运行这个:

var indexValues = new string[] { "apple fruit", "appricot", "ape", "avacado", "banana", "pear" };
var index = new LuceneAutoCompleteIndex("fn", 10);
index.AddRange(indexValues);

var matches = index.WhereStartsWith("app");
foreach (var match in matches)
{
    Console.WriteLine(match);
}

你明白了:

apple fruit
appricot