* merge search db and code removal from trunk..

- update_db operation is now %40+ faster...
- also search operations are faster...
This commit is contained in:
Faik Uygur
2007-03-01 21:13:37 +00:00
parent b508b05f55
commit c08d07d658
7 changed files with 20 additions and 272 deletions
+15 -26
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005, TUBITAK/UEKAE
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
@@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build
from pisi.metadata import MetaData
from pisi.files import Files
from pisi.file import File
import pisi.search
import pisi.lockeddbshelve as shelve
from pisi.version import Version
@@ -113,7 +112,6 @@ def init(database = True, write = True,
ctx.componentdb = pisi.component.ComponentDB()
ctx.packagedb = packagedb.init_db()
ctx.sourcedb = pisi.sourcedb.init()
pisi.search.init(['terms'], ['en', 'tr'])
else:
ctx.repodb = None
ctx.installdb = None
@@ -146,7 +144,6 @@ def finalize():
if ctx.sourcedb:
pisi.sourcedb.finalize()
ctx.sourcedb = None
pisi.search.finalize()
if ctx.dbenv:
ctx.dbenv.close()
ctx.dbenv_lock.close()
@@ -341,30 +338,22 @@ def info_name(package_name, installed=False):
files = None
return metadata, files
def search_package_names(query):
r = set()
packages = ctx.packagedb.list_packages()
for pkgname in packages:
if query in pkgname:
r.add(pkgname)
return r
def search_package_terms(terms, repo = pisi.itembyrepodb.all):
def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
r = pisi.search.query_terms('terms', lang, terms, repo = repo)
if search_names:
for term in terms:
r |= search_package_names(term)
return r
def search(package, term):
term = unicode(term).lower()
if term in unicode(package.name).lower() or \
term in unicode(package.summary).lower() or \
term in unicode(package.description).lower():
return True
def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
r = pisi.search.query('terms', lang, query, repo = repo)
if search_names:
r |= search_package_names(query)
return r
found = []
for name in ctx.packagedb.list_packages(repo):
pkg = ctx.packagedb.get_package(name, repo)
if terms == filter(lambda x:search(pkg, x), terms):
found.append(name)
return found
def check(package):
md, files = info(package, True)
+1 -9
View File
@@ -1510,14 +1510,6 @@ in summary, description, and package name fields.
default=False, help=_("Show details"))
self.parser.add_option_group(group)
def get_lang(self):
lang = ctx.get_option('language')
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
if not lang in ['en', 'tr']:
lang = 'en'
return lang
def run(self):
self.init(database = True, write = False)
@@ -1526,7 +1518,7 @@ in summary, description, and package name fields.
self.help()
return
r = pisi.api.search_package_terms(self.args, self.get_lang())
r = pisi.api.search_package_terms(self.args)
ctx.ui.info(_('%s packages found') % len(r))
ctx.config.options.short = not ctx.config.options.long
+4 -30
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005, TUBITAK/UEKAE
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
@@ -103,21 +103,7 @@ class PackageDB(object):
self.dr.add_item(dep_name, [ (name, dep) ], repo, txn)
# add component
ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn)
# index summary and description
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
ctx.txn_proc(proc, txn)
def clear(self, txn = None):
@@ -144,21 +130,9 @@ class PackageDB(object):
# all the list members are removed.
self.dr.remove_item(dep_name, repo, txn=txn)
# remove from component
ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn)
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
self.d.txn_proc(proc, txn)
def remove_repo(self, repo, txn = None):
-62
View File
@@ -1,62 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import pisi
import pisi.context as ctx
class Error(pisi.Error):
pass
class Exception(pisi.Exception):
pass
# API
from invertedindex import InvertedIndex
import preprocess as p
def init(ids, langs):
"initialize databases"
assert type(ids)==type([])
assert type(langs)==type([])
ctx.invidx = {}
for id in ids:
ctx.invidx[id] = {}
for lang in langs:
ctx.invidx[id][lang] = InvertedIndex(id, lang)
def finalize():
import pisi.context as ctx
if ctx.invidx:
for id in ctx.invidx.iterkeys():
for lang in ctx.invidx[id].iterkeys():
ctx.invidx[id][lang].close()
ctx.invidx = {}
def add_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn)
def remove_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn)
def query_terms(id, lang, terms, repo = None, txn = None):
terms = p.normalize(lang, terms)
return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn)
def query(id, lang, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
return query_terms(id, lang, terms, repo = repo, txn = txn)
-83
View File
@@ -1,83 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import types
import pisi.lockeddbshelve as shelve
from pisi.itembyrepodb import ItemByRepoDB
import pisi.itembyrepodb as itembyrepodb
class InvertedIndex(object):
"""a database of term -> set of documents"""
def __init__(self, id, lang):
self.d = ItemByRepoDB('ii-%s-%s' % (id, lang))
def close(self):
self.d.close()
def has_term(self, term, repo = None, txn = None):
return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn)
def get_term(self, term, repo = None, txn = None):
"""get set of doc ids given term"""
term = shelve.LockedDBShelf.encodekey(term)
def proc(txn):
if not self.has_term(term, repo=repo, txn=txn):
return set()
return self.d.get_item(term, repo=repo, txn=txn)
return self.d.txn_proc(proc, txn)
def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ):
"""get a union of all repository terms, not just the first repo in order.
get only basic repo info from the first repo"""
name = shelve.LockedDBShelf.encodekey(name)
def proc(txn):
terms= set()
if self.d.d.has_key(name):
s = self.d.d.get(name, txn=txn)
for repostr in self.d.order(repo = repo):
if s.has_key(repostr):
terms |= s[repostr]
return terms
return self.d.txn_proc(proc, txn)
def query(self, terms, repo = None, txn = None):
def proc(txn):
docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ]
if docs:
return reduce(lambda x,y: x.intersection(y), docs)
else:
return set()
return self.d.txn_proc(proc, txn)
def list_terms(self, repo = None, txn= None):
return self.d.list(f, repo=repo, txn=txn)
def add_doc(self, doc, terms, repo = None, txn = None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i, repo=repo, txn=txn)
term_i_docs.add(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
def remove_doc(self, doc, terms,repo=None, txn=None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i,repo=repo, txn=txn)
if doc in term_i_docs:
term_i_docs.remove(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
-30
View File
@@ -1,30 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import tokenize
import locale
def normalize(lang, terms):
if lang == "tr":
old_locale = locale.setlocale(locale.LC_CTYPE)
locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8")
terms = map(lambda x: unicode(x).lower(), terms)
if lang == "tr":
locale.setlocale(locale.LC_CTYPE, old_locale)
unique_terms = set()
for term in terms:
unique_terms.add(unicode(term))
return list(unique_terms)
def preprocess(lang, str):
terms = tokenize.tokenize(lang, str)
return normalize(lang, terms)
-32
View File
@@ -1,32 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import string
def tokenize(lang, str):
if type(str) != type(unicode()):
str = unicode(str)
sepchars = string.whitespace + string.punctuation
tokens = []
token = unicode()
for x in str:
if x in sepchars:
if len(token) > 0:
tokens.append(token)
token = unicode()
else:
token += x
if token:
tokens.append(token)
return tokens