2010-08-17 73 views
2

我是Lucene的新手,并试图对此进行分类。我索引是这样的:Lucene.net返回查询匹配的正确数量,但不是正确的文档

 Directory dir = FSDirectory.Open(new System.IO.DirectoryInfo(dirIndexDir)); 

     //Create the indexWriter 
     IndexWriter writer = new IndexWriter(dir, new StandardAnalyzer(Lucene.Net.Util.Version.LUCENE_29), true, 
      IndexWriter.MaxFieldLength.UNLIMITED); 


      Document doc = new Document(); 

      doc.Add(new Field("keyform_type", entry.keyForm.type, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
      doc.Add(new Field("keyform_lang", entry.keyForm.lang, Field.Store.YES, Field.Index.NOT_ANALYZED)); 

       doc.Add(new Field("keyform_dial", entry.keyForm.dial, Field.Store.YES, Field.Index.NOT_ANALYZED)); 

      doc.Add(new Field("keyform_reg", entry.keyForm.reg, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
      doc.Add(new Field("keyform_term", entry.keyForm.term.Value, Field.Store.YES, Field.Index.ANALYZED)); 

       if(entry.refForm.type!=null) 
        doc.Add(new Field("refform_type", entry.refForm.type, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       if(entry.refForm.lang!=null) 
        doc.Add(new Field("refform_lang", entry.refForm.lang, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       if (entry.refForm.dial != null) 
        doc.Add(new Field("refform_dial", entry.refForm.dial, Field.Store.YES, Field.Index.NOT_ANALYZED)); 

       if(entry.refForm.reg!=null) 
        doc.Add(new Field("refform_reg", entry.refForm.reg, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       if(entry.refForm.term.Value!=null) 
        doc.Add(new Field("refform_term", entry.refForm.term.Value, Field.Store.YES, Field.Index.ANALYZED)); 

       doc.Add(new Field("pos", entry.pos, Field.Store.YES, Field.Index.NOT_ANALYZED)); 

       for (int s = 0; s < entry.subject.Count; s++) 
       { 
        doc.Add(new Field("subject_"+s, entry.subject[s], Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       } 
       for (int g = 0; g < entry.sense.gloss.Count; g++) 
       { 
        doc.Add(new Field("gloss_"+g, entry.sense.gloss[g], Field.Store.YES, Field.Index.ANALYZED)); 

       } 
       if (entry.signature.action != null) 
        doc.Add(new Field("action", entry.signature.action, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       if (entry.signature.source != null) 
        doc.Add(new Field("source", entry.signature.source, Field.Store.YES, Field.Index.NOT_ANALYZED)); 
       if(entry.signature.date==0) 
        doc.Add(new Field("date", entry.signature.date.ToString(), Field.Store.YES, Field.Index.NOT_ANALYZED)); 
      //Add the doc 
      writer.AddDocument(doc); 

     writer.Close(); 

我那么查询使用此代码:

 //Doesn't matter what term is, same result 
     string term="workers"; 

     Directory dir = FSDirectory.Open(new System.IO.DirectoryInfo(luceneDir)); 

     IndexSearcher searcher = new IndexSearcher(dir, true); 
     List<string> b=new List<string>(); 
     b.Add("keyform_gloss"); 
     b.Add("keyform_term"); 
     b.Add("refform_term"); 
     b.Add("refform_gloss"); 
     for (int i = 0; i < nMaxDupes; i++) 
      b.Add("gloss_" + i.ToString()); 
     MultiFieldQueryParser mfqp = new MultiFieldQueryParser(Lucene.Net.Util.Version.LUCENE_29, 
      b.ToArray(), new StandardAnalyzer()); 
     Query q = mfqp.Parse(term); 
     TopDocs td = searcher.Search(q, 300); 

     for (int i = 0; i < td.totalHits; i++) 
     { 
      //Generate a dictionaryEntry for each hit 
      Document doc = searcher.Doc(i); 

      //Access the document fields, blah 
     } 

不管是什么项的值,返回的Lucene索引中,其中X第X文档=实际符合条款的文件数量。当我使用LUKE浏览索引时,相同的手动查询(keyform_term:term gloss_0:term等)会返回正确数量的结果以及与这些结果匹配的正确文档。

但是,上面的C#代码总是返回第一个X文档,它们不一定包含任何搜索字段中的搜索词。他们甚至没有接近。

我在做什么错?我知道索引是好的,因为我可以在LUKE中搜索它,所以它必须是在查询中的东西...

谢谢!

回答

6

行:

Document doc = searcher.Doc(i); 

应该

Document doc = searcher.Doc(td.scoreDocs[i].doc); 

或正确的C#语法当量(我是一个Java的家伙,抱歉)

+0

恰好就是这样,谢谢!如果可以的话,我会投票回答。 – 2010-08-17 20:43:56

+0

只需单击复选标记以将其标记为正确的答案。 ;) – bajafresh4life 2010-08-18 02:27:52

+0

哦......谢谢! :) – 2010-08-18 17:52:08