from org.apache.lucene.document import Field, Document
from org.apache.lucene.analysis.standard import StandardAnalyzer
from org.apache.lucene.index import IndexWriter
from org.apache.lucene.store import RAMDirectory
from org.apache.lucene.search import IndexSearcher,Query
from org.apache.lucene.queryParser import QueryParser
# create index file in memory, optimize, and perform searches
text=["Use dmelt data analysis for any statistical data analysis. DMelt is good for this",
"Dmelt is event more. You can analyze text, images and use in natural sciences",
"Any activity can be possible. Any combination, any task. Try it to see this",
"Let those who are in favour with their stars, Of public honour and proud titles boast",
"Whilst I, whom fortune of such triumph bars, Unlook for joy in that I honour most"]
def addText(doc,txt):
doc.add(Field("content",txt, Field.Store.YES, Field.Index.TOKENIZED, Field.TermVector.WITH_POSITIONS))
writer.addDocument(doc)
writer.optimize()
print "Create index in memory using input string"
xdir = RAMDirectory()
writer = IndexWriter(xdir,StandardAnalyzer(),True)
doc=Document()
for t in text: addText(doc,t) # add text page by page
writer.close()
print 'Find the pattern inside text..'
searcher = IndexSearcher(xdir)
parser = QueryParser("content", StandardAnalyzer())
query=parser.parse("dmelt") # search dmelt
hits = searcher.search(query)
print "Searching for: ",query.toString("content")
print "Number of found hits=",hits.length()," Look at score for each page:"
for i in range(hits.length()):
print "doc=",hits.id(i)," score=",hits.score(i)
Ads help maintain this website.