* merge search db and code removal from trunk..

- update_db operation is now %40+ faster...
- also search operations are faster...
This commit is contained in:
Faik Uygur
2007-03-01 21:13:37 +00:00
parent b508b05f55
commit c08d07d658
7 changed files with 20 additions and 272 deletions
+15 -26
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# #
# Copyright (C) 2005, TUBITAK/UEKAE # Copyright (C) 2005 - 2007, TUBITAK/UEKAE
# #
# This program is free software; you can redistribute it and/or modify it under # This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free # the terms of the GNU General Public License as published by the Free
@@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build
from pisi.metadata import MetaData from pisi.metadata import MetaData
from pisi.files import Files from pisi.files import Files
from pisi.file import File from pisi.file import File
import pisi.search
import pisi.lockeddbshelve as shelve import pisi.lockeddbshelve as shelve
from pisi.version import Version from pisi.version import Version
@@ -113,7 +112,6 @@ def init(database = True, write = True,
ctx.componentdb = pisi.component.ComponentDB() ctx.componentdb = pisi.component.ComponentDB()
ctx.packagedb = packagedb.init_db() ctx.packagedb = packagedb.init_db()
ctx.sourcedb = pisi.sourcedb.init() ctx.sourcedb = pisi.sourcedb.init()
pisi.search.init(['terms'], ['en', 'tr'])
else: else:
ctx.repodb = None ctx.repodb = None
ctx.installdb = None ctx.installdb = None
@@ -146,7 +144,6 @@ def finalize():
if ctx.sourcedb: if ctx.sourcedb:
pisi.sourcedb.finalize() pisi.sourcedb.finalize()
ctx.sourcedb = None ctx.sourcedb = None
pisi.search.finalize()
if ctx.dbenv: if ctx.dbenv:
ctx.dbenv.close() ctx.dbenv.close()
ctx.dbenv_lock.close() ctx.dbenv_lock.close()
@@ -341,30 +338,22 @@ def info_name(package_name, installed=False):
files = None files = None
return metadata, files return metadata, files
def search_package_names(query): def search_package_terms(terms, repo = pisi.itembyrepodb.all):
r = set()
packages = ctx.packagedb.list_packages()
for pkgname in packages:
if query in pkgname:
r.add(pkgname)
return r
def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all): def search(package, term):
if not lang: term = unicode(term).lower()
lang = pisi.pxml.autoxml.LocalText.get_lang() if term in unicode(package.name).lower() or \
r = pisi.search.query_terms('terms', lang, terms, repo = repo) term in unicode(package.summary).lower() or \
if search_names: term in unicode(package.description).lower():
for term in terms: return True
r |= search_package_names(term)
return r
def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all): found = []
if not lang: for name in ctx.packagedb.list_packages(repo):
lang = pisi.pxml.autoxml.LocalText.get_lang() pkg = ctx.packagedb.get_package(name, repo)
r = pisi.search.query('terms', lang, query, repo = repo) if terms == filter(lambda x:search(pkg, x), terms):
if search_names: found.append(name)
r |= search_package_names(query)
return r return found
def check(package): def check(package):
md, files = info(package, True) md, files = info(package, True)
+1 -9
View File
@@ -1510,14 +1510,6 @@ in summary, description, and package name fields.
default=False, help=_("Show details")) default=False, help=_("Show details"))
self.parser.add_option_group(group) self.parser.add_option_group(group)
def get_lang(self):
lang = ctx.get_option('language')
if not lang:
lang = pisi.pxml.autoxml.LocalText.get_lang()
if not lang in ['en', 'tr']:
lang = 'en'
return lang
def run(self): def run(self):
self.init(database = True, write = False) self.init(database = True, write = False)
@@ -1526,7 +1518,7 @@ in summary, description, and package name fields.
self.help() self.help()
return return
r = pisi.api.search_package_terms(self.args, self.get_lang()) r = pisi.api.search_package_terms(self.args)
ctx.ui.info(_('%s packages found') % len(r)) ctx.ui.info(_('%s packages found') % len(r))
ctx.config.options.short = not ctx.config.options.long ctx.config.options.short = not ctx.config.options.long
+4 -30
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# #
# Copyright (C) 2005, TUBITAK/UEKAE # Copyright (C) 2005 - 2007, TUBITAK/UEKAE
# #
# This program is free software; you can redistribute it and/or modify it under # This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free # the terms of the GNU General Public License as published by the Free
@@ -103,21 +103,7 @@ class PackageDB(object):
self.dr.add_item(dep_name, [ (name, dep) ], repo, txn) self.dr.add_item(dep_name, [ (name, dep) ], repo, txn)
# add component # add component
ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn) ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn)
# index summary and description
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
ctx.txn_proc(proc, txn) ctx.txn_proc(proc, txn)
def clear(self, txn = None): def clear(self, txn = None):
@@ -144,21 +130,9 @@ class PackageDB(object):
# all the list members are removed. # all the list members are removed.
self.dr.remove_item(dep_name, repo, txn=txn) self.dr.remove_item(dep_name, repo, txn=txn)
# remove from component
ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn) ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn)
search_keys = {}
for (lang, doc) in package_info.summary.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for (lang, doc) in package_info.description.iteritems():
text = search_keys.get(lang, "")
text += " %s" % doc
search_keys[lang] = text
for lang in search_keys:
text = search_keys[lang]
# FIXME: other languages should be searchable too
if lang in ('en', 'tr'):
pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
self.d.txn_proc(proc, txn) self.d.txn_proc(proc, txn)
def remove_repo(self, repo, txn = None): def remove_repo(self, repo, txn = None):
-62
View File
@@ -1,62 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import pisi
import pisi.context as ctx
class Error(pisi.Error):
pass
class Exception(pisi.Exception):
pass
# API
from invertedindex import InvertedIndex
import preprocess as p
def init(ids, langs):
"initialize databases"
assert type(ids)==type([])
assert type(langs)==type([])
ctx.invidx = {}
for id in ids:
ctx.invidx[id] = {}
for lang in langs:
ctx.invidx[id][lang] = InvertedIndex(id, lang)
def finalize():
import pisi.context as ctx
if ctx.invidx:
for id in ctx.invidx.iterkeys():
for lang in ctx.invidx[id].iterkeys():
ctx.invidx[id][lang].close()
ctx.invidx = {}
def add_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn)
def remove_doc(id, lang, docid, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn)
def query_terms(id, lang, terms, repo = None, txn = None):
terms = p.normalize(lang, terms)
return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn)
def query(id, lang, str, repo = None, txn = None):
terms = p.preprocess(lang, str)
return query_terms(id, lang, terms, repo = repo, txn = txn)
-83
View File
@@ -1,83 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import types
import pisi.lockeddbshelve as shelve
from pisi.itembyrepodb import ItemByRepoDB
import pisi.itembyrepodb as itembyrepodb
class InvertedIndex(object):
"""a database of term -> set of documents"""
def __init__(self, id, lang):
self.d = ItemByRepoDB('ii-%s-%s' % (id, lang))
def close(self):
self.d.close()
def has_term(self, term, repo = None, txn = None):
return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn)
def get_term(self, term, repo = None, txn = None):
"""get set of doc ids given term"""
term = shelve.LockedDBShelf.encodekey(term)
def proc(txn):
if not self.has_term(term, repo=repo, txn=txn):
return set()
return self.d.get_item(term, repo=repo, txn=txn)
return self.d.txn_proc(proc, txn)
def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ):
"""get a union of all repository terms, not just the first repo in order.
get only basic repo info from the first repo"""
name = shelve.LockedDBShelf.encodekey(name)
def proc(txn):
terms= set()
if self.d.d.has_key(name):
s = self.d.d.get(name, txn=txn)
for repostr in self.d.order(repo = repo):
if s.has_key(repostr):
terms |= s[repostr]
return terms
return self.d.txn_proc(proc, txn)
def query(self, terms, repo = None, txn = None):
def proc(txn):
docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ]
if docs:
return reduce(lambda x,y: x.intersection(y), docs)
else:
return set()
return self.d.txn_proc(proc, txn)
def list_terms(self, repo = None, txn= None):
return self.d.list(f, repo=repo, txn=txn)
def add_doc(self, doc, terms, repo = None, txn = None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i, repo=repo, txn=txn)
term_i_docs.add(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
def remove_doc(self, doc, terms,repo=None, txn=None):
def f(txn):
for term_i in terms:
term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i,repo=repo, txn=txn)
if doc in term_i_docs:
term_i_docs.remove(doc)
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
return self.d.txn_proc(f, txn)
-30
View File
@@ -1,30 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import tokenize
import locale
def normalize(lang, terms):
if lang == "tr":
old_locale = locale.setlocale(locale.LC_CTYPE)
locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8")
terms = map(lambda x: unicode(x).lower(), terms)
if lang == "tr":
locale.setlocale(locale.LC_CTYPE, old_locale)
unique_terms = set()
for term in terms:
unique_terms.add(unicode(term))
return list(unique_terms)
def preprocess(lang, str):
terms = tokenize.tokenize(lang, str)
return normalize(lang, terms)
-32
View File
@@ -1,32 +0,0 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2005-2006, TUBITAK/UEKAE
#
# This program is free software; you can redistribute it and/or modify it under
# the terms of the GNU General Public License as published by the Free
# Software Foundation; either version 2 of the License, or (at your option)
# any later version.
#
# Please read the COPYING file.
#
import string
def tokenize(lang, str):
if type(str) != type(unicode()):
str = unicode(str)
sepchars = string.whitespace + string.punctuation
tokens = []
token = unicode()
for x in str:
if x in sepchars:
if len(token) > 0:
tokens.append(token)
token = unicode()
else:
token += x
if token:
tokens.append(token)
return tokens