Oh, you will love this one...

The DB 101 seems to be, not worked here very well

All the search phantasmogoria db code gibbed and replaced
by 15 lines of code...

+ update_repo increased %30-40 percent...
+ you can now search any keyword _within_ a word...
+ multi term search speed also increased
+ decreased code size

- nothing really... just some turkish locale lowering stuff in some 
rare cases which was also fixed by Gurer in the old code not merged
but may also be added here some time in the future...

oops and special thanks goes to Gurer and Mehmet...
This commit is contained in:
Faik Uygur
2007-03-01 13:14:29 +00:00
parent a6bfb640c5
commit 6f711f0723
7 changed files with 16 additions and 271 deletions
+14 -25
View File
@@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build
from pisi.metadata import MetaData
from pisi.files import Files
from pisi.file import File
import pisi.search
import pisi.lockeddbshelve as shelve
from pisi.version import Version
@@ -113,7 +112,6 @@ def init(database = True, write = True,
ctx.componentdb = pisi.component.ComponentDB()
ctx.packagedb = packagedb.init_db()
ctx.sourcedb = pisi.sourcedb.init()
pisi.search.init(['terms'], ['en', 'tr'])
else:
ctx.repodb = None
ctx.installdb = None
@@ -146,7 +144,6 @@ def finalize():
if ctx.sourcedb:
pisi.sourcedb.finalize()
ctx.sourcedb = None
pisi.search.finalize()
if ctx.dbenv:
ctx.dbenv.close()
ctx.dbenv_lock.close()
@@ -341,30 +338,22 @@ def info_name(package_name, installed=False):
files = None
return metadata, files
def search_package_names(query):
r = set()
packages = ctx.packagedb.list_packages()
for pkgname in packages:
if query in pkgname:
r.add(pkgname)
return r
def search_package_terms(terms, repo = pisi.itembyrepodb.all):
def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
r = pisi.search.query_terms('terms', lang, terms, repo = repo)
if search_names:
for term in terms:
r |= search_package_names(term)
return r
def search(package, term):
term = unicode(term).lower()
if term in unicode(package.name).lower() or \
term in unicode(package.summary).lower() or \
term in unicode(package.description).lower():
return True
def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
r = pisi.search.query('terms', lang, query, repo = repo)
if search_names:
r |= search_package_names(query)
return r
found = []
for name in ctx.packagedb.list_packages(repo):
pkg = ctx.packagedb.get_package(name, repo)
if terms == filter(lambda x:search(pkg, x), terms):
found.append(name)
return found
def check(package):
md, files = info(package, True)
+1 -9
View File
@@ -1567,14 +1567,6 @@ in summary, description, and package name fields.
default=False, help=_("Show details"))
self.parser.add_option_group(group)
def get_lang(self):
lang = ctx.get_option('language')
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
if not lang in ['en', 'tr']:
lang = 'en'
return lang
def run(self):
self.init(database = True, write = False)
@@ -1583,7 +1575,7 @@ in summary, description, and package name fields.
self.help()
return
r = pisi.api.search_package_terms(self.args, self.get_lang())
r = pisi.api.search_package_terms(self.args)
ctx.ui.info(_('%s packages found') % len(r))
ctx.config.options.short = not ctx.config.options.long
+1 -30
View File
@@ -103,21 +103,7 @@ class PackageDB(object):
self.dr.add_item(dep_name, [ (name, dep) ], repo, txn)
# add component
ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn)
# index summary and description
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
ctx.txn_proc(proc, txn)
def clear(self, txn = None):
@@ -144,21 +130,6 @@ class PackageDB(object):
# all the list members are removed.
self.dr.remove_item(dep_name, repo, txn=txn)
ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn)
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
self.d.txn_proc(proc, txn)
def remove_repo(self, repo, txn = None):
-62
View File
@@ -1,62 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import pisi
import pisi.context as ctx
class Error(pisi.Error):
pass
class Exception(pisi.Exception):
pass
# API
from invertedindex import InvertedIndex
import preprocess as p
def init(ids, langs):
"initialize databases"
assert type(ids)==type([])
assert type(langs)==type([])
ctx.invidx = {}
for id in ids:
ctx.invidx[id] = {}
for lang in langs:
ctx.invidx[id][lang] = InvertedIndex(id, lang)
def finalize():
import pisi.context as ctx
if ctx.invidx:
for id in ctx.invidx.iterkeys():
for lang in ctx.invidx[id].iterkeys():
ctx.invidx[id][lang].close()
ctx.invidx = {}
def add_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn)
def remove_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn)
def query_terms(id, lang, terms, repo = None, txn = None):
terms = p.normalize(lang, terms)
return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn)
def query(id, lang, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
return query_terms(id, lang, terms, repo = repo, txn = txn)
-83
View File
@@ -1,83 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import types
import pisi.lockeddbshelve as shelve
from pisi.itembyrepodb import ItemByRepoDB
import pisi.itembyrepodb as itembyrepodb
class InvertedIndex(object):
"""a database of term -> set of documents"""
def __init__(self, id, lang):
self.d = ItemByRepoDB('ii-%s-%s' % (id, lang))
def close(self):
self.d.close()
def has_term(self, term, repo = None, txn = None):
return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn)
def get_term(self, term, repo = None, txn = None):
"""get set of doc ids given term"""
term = shelve.LockedDBShelf.encodekey(term)
def proc(txn):
if not self.has_term(term, repo=repo, txn=txn):
return set()
return self.d.get_item(term, repo=repo, txn=txn)
return self.d.txn_proc(proc, txn)
def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ):
"""get a union of all repository terms, not just the first repo in order.
get only basic repo info from the first repo"""
name = shelve.LockedDBShelf.encodekey(name)
def proc(txn):
terms= set()
if self.d.d.has_key(name):
s = self.d.d.get(name, txn=txn)
for repostr in self.d.order(repo = repo):
if s.has_key(repostr):
terms |= s[repostr]
return terms
return self.d.txn_proc(proc, txn)
def query(self, terms, repo = None, txn = None):
def proc(txn):
docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ]
if docs:
return reduce(lambda x,y: x.intersection(y), docs)
else:
return set()
return self.d.txn_proc(proc, txn)
def list_terms(self, repo = None, txn= None):
return self.d.list(f, repo=repo, txn=txn)
def add_doc(self, doc, terms, repo = None, txn = None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i, repo=repo, txn=txn)
term_i_docs.add(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
def remove_doc(self, doc, terms,repo=None, txn=None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i,repo=repo, txn=txn)
if doc in term_i_docs:
term_i_docs.remove(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
-30
View File
@@ -1,30 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import tokenize
import locale
def normalize(lang, terms):
if lang == "tr":
old_locale = locale.setlocale(locale.LC_CTYPE)
locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8")
terms = map(lambda x: unicode(x).lower(), terms)
if lang == "tr":
locale.setlocale(locale.LC_CTYPE, old_locale)
unique_terms = set()
for term in terms:
unique_terms.add(unicode(term))
return list(unique_terms)
def preprocess(lang, str):
terms = tokenize.tokenize(lang, str)
return normalize(lang, terms)
-32
View File
@@ -1,32 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import string
def tokenize(lang, str):
if type(str) != type(unicode()):
str = unicode(str)
sepchars = string.whitespace + string.punctuation
tokens = []
token = unicode()
for x in str:
if x in sepchars:
if len(token) > 0:
tokens.append(token)
token = unicode()
else:
token += x
if token:
tokens.append(token)
return tokens