Searching a text in a document using Apache Lucene
Code: "lucene_search_memory.py". Programming language: Python DMelt Version 1.9. Last modified: 08/21/2017. License: Pro
https://datamelt.org/code/cache/lucene_search_memory_820.py
To run this script using the DMelt IDE, copy the above URL link to the menu [File]→[Read script from URL] of the DMelt IDE.


from org.apache.lucene.document import Field, Document
from org.apache.lucene.analysis.standard import StandardAnalyzer
from org.apache.lucene.index import IndexWriter
from org.apache.lucene.store import RAMDirectory
from org.apache.lucene.search import IndexSearcher,Query
from org.apache.lucene.queryParser import QueryParser


# create index file in memory, optimize, and perform searches
text=["Use dmelt data analysis for any statistical data analysis. DMelt is good for this",
     "Dmelt is event more. You can analyze text, images and use in natural sciences", 
     "Any activity can be possible. Any combination, any task. Try it to see this",
     "Let those who are in favour with their stars, Of public honour and proud titles boast",
     "Whilst I, whom fortune of such triumph bars, Unlook for joy in that I honour most"]

def addText(doc,txt):
  doc.add(Field("content",txt, Field.Store.YES, Field.Index.TOKENIZED, Field.TermVector.WITH_POSITIONS))
  writer.addDocument(doc)
  writer.optimize()


print "Create index in memory using input string"
xdir = RAMDirectory()
writer = IndexWriter(xdir,StandardAnalyzer(),True)
doc=Document()
for t in text: addText(doc,t) # add text page by page 
writer.close()


print 'Find the pattern inside text..'
searcher = IndexSearcher(xdir)
parser = QueryParser("content", StandardAnalyzer())
query=parser.parse("dmelt") # search dmelt 
hits = searcher.search(query)
print "Searching for: ",query.toString("content")
print "Number of found hits=",hits.length()," Look at score for each page:" 
for i in range(hits.length()):
    print "doc=",hits.id(i)," score=",hits.score(i)



Ads help maintain this website.