From c08d07d658f9fae697e822c66edaf42c61fd0549 Mon Sep 17 00:00:00 2001 From: Faik Uygur Date: Thu, 1 Mar 2007 21:13:37 +0000 Subject: [PATCH] * merge search db and code removal from trunk.. - update_db operation is now %40+ faster... - also search operations are faster... --- pisi/api.py | 41 +++++++----------- pisi/cli/commands.py | 10 +---- pisi/packagedb.py | 34 ++------------- pisi/search/__init__.py | 62 --------------------------- pisi/search/invertedindex.py | 83 ------------------------------------ pisi/search/preprocess.py | 30 ------------- pisi/search/tokenize.py | 32 -------------- 7 files changed, 20 insertions(+), 272 deletions(-) delete mode 100644 pisi/search/__init__.py delete mode 100644 pisi/search/invertedindex.py delete mode 100644 pisi/search/preprocess.py delete mode 100644 pisi/search/tokenize.py diff --git a/pisi/api.py b/pisi/api.py index 1a6b77d7..9bc3a31c 100644 --- a/pisi/api.py +++ b/pisi/api.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- # -# Copyright (C) 2005, TUBITAK/UEKAE +# Copyright (C) 2005 - 2007, TUBITAK/UEKAE # # This program is free software; you can redistribute it and/or modify it under # the terms of the GNU General Public License as published by the Free @@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build from pisi.metadata import MetaData from pisi.files import Files from pisi.file import File -import pisi.search import pisi.lockeddbshelve as shelve from pisi.version import Version @@ -113,7 +112,6 @@ def init(database = True, write = True, ctx.componentdb = pisi.component.ComponentDB() ctx.packagedb = packagedb.init_db() ctx.sourcedb = pisi.sourcedb.init() - pisi.search.init(['terms'], ['en', 'tr']) else: ctx.repodb = None ctx.installdb = None @@ -146,7 +144,6 @@ def finalize(): if ctx.sourcedb: pisi.sourcedb.finalize() ctx.sourcedb = None - pisi.search.finalize() if ctx.dbenv: ctx.dbenv.close() ctx.dbenv_lock.close() @@ -341,30 +338,22 @@ def info_name(package_name, installed=False): files = None return metadata, files -def search_package_names(query): - r = set() - packages = ctx.packagedb.list_packages() - for pkgname in packages: - if query in pkgname: - r.add(pkgname) - return r +def search_package_terms(terms, repo = pisi.itembyrepodb.all): -def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all): - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - r = pisi.search.query_terms('terms', lang, terms, repo = repo) - if search_names: - for term in terms: - r |= search_package_names(term) - return r + def search(package, term): + term = unicode(term).lower() + if term in unicode(package.name).lower() or \ + term in unicode(package.summary).lower() or \ + term in unicode(package.description).lower(): + return True -def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all): - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - r = pisi.search.query('terms', lang, query, repo = repo) - if search_names: - r |= search_package_names(query) - return r + found = [] + for name in ctx.packagedb.list_packages(repo): + pkg = ctx.packagedb.get_package(name, repo) + if terms == filter(lambda x:search(pkg, x), terms): + found.append(name) + + return found def check(package): md, files = info(package, True) diff --git a/pisi/cli/commands.py b/pisi/cli/commands.py index 53374e56..c25db05d 100644 --- a/pisi/cli/commands.py +++ b/pisi/cli/commands.py @@ -1510,14 +1510,6 @@ in summary, description, and package name fields. default=False, help=_("Show details")) self.parser.add_option_group(group) - def get_lang(self): - lang = ctx.get_option('language') - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - if not lang in ['en', 'tr']: - lang = 'en' - return lang - def run(self): self.init(database = True, write = False) @@ -1526,7 +1518,7 @@ in summary, description, and package name fields. self.help() return - r = pisi.api.search_package_terms(self.args, self.get_lang()) + r = pisi.api.search_package_terms(self.args) ctx.ui.info(_('%s packages found') % len(r)) ctx.config.options.short = not ctx.config.options.long diff --git a/pisi/packagedb.py b/pisi/packagedb.py index 736f6d5b..abc569fd 100644 --- a/pisi/packagedb.py +++ b/pisi/packagedb.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- # -# Copyright (C) 2005, TUBITAK/UEKAE +# Copyright (C) 2005 - 2007, TUBITAK/UEKAE # # This program is free software; you can redistribute it and/or modify it under # the terms of the GNU General Public License as published by the Free @@ -103,21 +103,7 @@ class PackageDB(object): self.dr.add_item(dep_name, [ (name, dep) ], repo, txn) # add component ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn) - # index summary and description - search_keys = {} - for (lang, doc) in package_info.summary.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for (lang, doc) in package_info.description.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for lang in search_keys: - text = search_keys[lang] - # FIXME: other languages should be searchable too - if lang in ('en', 'tr'): - pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn) + ctx.txn_proc(proc, txn) def clear(self, txn = None): @@ -144,21 +130,9 @@ class PackageDB(object): # all the list members are removed. self.dr.remove_item(dep_name, repo, txn=txn) + # remove from component ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn) - search_keys = {} - for (lang, doc) in package_info.summary.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for (lang, doc) in package_info.description.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for lang in search_keys: - text = search_keys[lang] - # FIXME: other languages should be searchable too - if lang in ('en', 'tr'): - pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn) + self.d.txn_proc(proc, txn) def remove_repo(self, repo, txn = None): diff --git a/pisi/search/__init__.py b/pisi/search/__init__.py deleted file mode 100644 index cf5a099b..00000000 --- a/pisi/search/__init__.py +++ /dev/null @@ -1,62 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import pisi -import pisi.context as ctx - -class Error(pisi.Error): - pass - -class Exception(pisi.Exception): - pass - -# API - -from invertedindex import InvertedIndex -import preprocess as p - -def init(ids, langs): - "initialize databases" - - assert type(ids)==type([]) - assert type(langs)==type([]) - - ctx.invidx = {} - for id in ids: - ctx.invidx[id] = {} - for lang in langs: - ctx.invidx[id][lang] = InvertedIndex(id, lang) - -def finalize(): - import pisi.context as ctx - - if ctx.invidx: - for id in ctx.invidx.iterkeys(): - for lang in ctx.invidx[id].iterkeys(): - ctx.invidx[id][lang].close() - ctx.invidx = {} - -def add_doc(id, lang, docid, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn) - -def remove_doc(id, lang, docid, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn) - -def query_terms(id, lang, terms, repo = None, txn = None): - terms = p.normalize(lang, terms) - return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn) - -def query(id, lang, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - return query_terms(id, lang, terms, repo = repo, txn = txn) diff --git a/pisi/search/invertedindex.py b/pisi/search/invertedindex.py deleted file mode 100644 index b98158ec..00000000 --- a/pisi/search/invertedindex.py +++ /dev/null @@ -1,83 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import types - -import pisi.lockeddbshelve as shelve -from pisi.itembyrepodb import ItemByRepoDB -import pisi.itembyrepodb as itembyrepodb - -class InvertedIndex(object): - """a database of term -> set of documents""" - - def __init__(self, id, lang): - self.d = ItemByRepoDB('ii-%s-%s' % (id, lang)) - - def close(self): - self.d.close() - - def has_term(self, term, repo = None, txn = None): - return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn) - - def get_term(self, term, repo = None, txn = None): - """get set of doc ids given term""" - term = shelve.LockedDBShelf.encodekey(term) - def proc(txn): - if not self.has_term(term, repo=repo, txn=txn): - return set() - return self.d.get_item(term, repo=repo, txn=txn) - return self.d.txn_proc(proc, txn) - - def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ): - """get a union of all repository terms, not just the first repo in order. - get only basic repo info from the first repo""" - name = shelve.LockedDBShelf.encodekey(name) - def proc(txn): - terms= set() - if self.d.d.has_key(name): - s = self.d.d.get(name, txn=txn) - for repostr in self.d.order(repo = repo): - if s.has_key(repostr): - terms |= s[repostr] - return terms - return self.d.txn_proc(proc, txn) - - def query(self, terms, repo = None, txn = None): - def proc(txn): - docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ] - if docs: - return reduce(lambda x,y: x.intersection(y), docs) - else: - return set() - return self.d.txn_proc(proc, txn) - - def list_terms(self, repo = None, txn= None): - return self.d.list(f, repo=repo, txn=txn) - - def add_doc(self, doc, terms, repo = None, txn = None): - def f(txn): - for term_i in terms: - term_i = shelve.LockedDBShelf.encodekey(term_i) - term_i_docs = self.get_term(term_i, repo=repo, txn=txn) - term_i_docs.add(doc) - self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update - return self.d.txn_proc(f, txn) - - def remove_doc(self, doc, terms,repo=None, txn=None): - def f(txn): - for term_i in terms: - term_i = shelve.LockedDBShelf.encodekey(term_i) - term_i_docs = self.get_term(term_i,repo=repo, txn=txn) - if doc in term_i_docs: - term_i_docs.remove(doc) - self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update - return self.d.txn_proc(f, txn) diff --git a/pisi/search/preprocess.py b/pisi/search/preprocess.py deleted file mode 100644 index 97964091..00000000 --- a/pisi/search/preprocess.py +++ /dev/null @@ -1,30 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import tokenize -import locale - -def normalize(lang, terms): - if lang == "tr": - old_locale = locale.setlocale(locale.LC_CTYPE) - locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8") - terms = map(lambda x: unicode(x).lower(), terms) - if lang == "tr": - locale.setlocale(locale.LC_CTYPE, old_locale) - unique_terms = set() - for term in terms: - unique_terms.add(unicode(term)) - return list(unique_terms) - -def preprocess(lang, str): - terms = tokenize.tokenize(lang, str) - return normalize(lang, terms) diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py deleted file mode 100644 index 0023e45a..00000000 --- a/pisi/search/tokenize.py +++ /dev/null @@ -1,32 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import string - -def tokenize(lang, str): - if type(str) != type(unicode()): - str = unicode(str) - sepchars = string.whitespace + string.punctuation - tokens = [] - token = unicode() - for x in str: - if x in sepchars: - if len(token) > 0: - tokens.append(token) - token = unicode() - else: - token += x - - if token: - tokens.append(token) - - return tokens