From 6f711f07231710251dfe5447edbfd5fb49737787 Mon Sep 17 00:00:00 2001 From: Faik Uygur Date: Thu, 1 Mar 2007 13:14:29 +0000 Subject: [PATCH] Oh, you will love this one... The DB 101 seems to be, not worked here very well All the search phantasmogoria db code gibbed and replaced by 15 lines of code... + update_repo increased %30-40 percent... + you can now search any keyword _within_ a word... + multi term search speed also increased + decreased code size - nothing really... just some turkish locale lowering stuff in some rare cases which was also fixed by Gurer in the old code not merged but may also be added here some time in the future... oops and special thanks goes to Gurer and Mehmet... --- pisi/api.py | 39 ++++++----------- pisi/cli/commands.py | 10 +---- pisi/packagedb.py | 31 +------------- pisi/search/__init__.py | 62 --------------------------- pisi/search/invertedindex.py | 83 ------------------------------------ pisi/search/preprocess.py | 30 ------------- pisi/search/tokenize.py | 32 -------------- 7 files changed, 16 insertions(+), 271 deletions(-) delete mode 100644 pisi/search/__init__.py delete mode 100644 pisi/search/invertedindex.py delete mode 100644 pisi/search/preprocess.py delete mode 100644 pisi/search/tokenize.py diff --git a/pisi/api.py b/pisi/api.py index c11e159e..9bc3a31c 100644 --- a/pisi/api.py +++ b/pisi/api.py @@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build from pisi.metadata import MetaData from pisi.files import Files from pisi.file import File -import pisi.search import pisi.lockeddbshelve as shelve from pisi.version import Version @@ -113,7 +112,6 @@ def init(database = True, write = True, ctx.componentdb = pisi.component.ComponentDB() ctx.packagedb = packagedb.init_db() ctx.sourcedb = pisi.sourcedb.init() - pisi.search.init(['terms'], ['en', 'tr']) else: ctx.repodb = None ctx.installdb = None @@ -146,7 +144,6 @@ def finalize(): if ctx.sourcedb: pisi.sourcedb.finalize() ctx.sourcedb = None - pisi.search.finalize() if ctx.dbenv: ctx.dbenv.close() ctx.dbenv_lock.close() @@ -341,30 +338,22 @@ def info_name(package_name, installed=False): files = None return metadata, files -def search_package_names(query): - r = set() - packages = ctx.packagedb.list_packages() - for pkgname in packages: - if query in pkgname: - r.add(pkgname) - return r +def search_package_terms(terms, repo = pisi.itembyrepodb.all): -def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all): - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - r = pisi.search.query_terms('terms', lang, terms, repo = repo) - if search_names: - for term in terms: - r |= search_package_names(term) - return r + def search(package, term): + term = unicode(term).lower() + if term in unicode(package.name).lower() or \ + term in unicode(package.summary).lower() or \ + term in unicode(package.description).lower(): + return True -def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all): - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - r = pisi.search.query('terms', lang, query, repo = repo) - if search_names: - r |= search_package_names(query) - return r + found = [] + for name in ctx.packagedb.list_packages(repo): + pkg = ctx.packagedb.get_package(name, repo) + if terms == filter(lambda x:search(pkg, x), terms): + found.append(name) + + return found def check(package): md, files = info(package, True) diff --git a/pisi/cli/commands.py b/pisi/cli/commands.py index 2d61c5d7..79faaa79 100644 --- a/pisi/cli/commands.py +++ b/pisi/cli/commands.py @@ -1567,14 +1567,6 @@ in summary, description, and package name fields. default=False, help=_("Show details")) self.parser.add_option_group(group) - def get_lang(self): - lang = ctx.get_option('language') - if not lang: - lang = pisi.pxml.autoxml.LocalText.get_lang() - if not lang in ['en', 'tr']: - lang = 'en' - return lang - def run(self): self.init(database = True, write = False) @@ -1583,7 +1575,7 @@ in summary, description, and package name fields. self.help() return - r = pisi.api.search_package_terms(self.args, self.get_lang()) + r = pisi.api.search_package_terms(self.args) ctx.ui.info(_('%s packages found') % len(r)) ctx.config.options.short = not ctx.config.options.long diff --git a/pisi/packagedb.py b/pisi/packagedb.py index cb650c0b..f0ae84d2 100644 --- a/pisi/packagedb.py +++ b/pisi/packagedb.py @@ -103,21 +103,7 @@ class PackageDB(object): self.dr.add_item(dep_name, [ (name, dep) ], repo, txn) # add component ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn) - # index summary and description - search_keys = {} - for (lang, doc) in package_info.summary.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for (lang, doc) in package_info.description.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for lang in search_keys: - text = search_keys[lang] - # FIXME: other languages should be searchable too - if lang in ('en', 'tr'): - pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn) + ctx.txn_proc(proc, txn) def clear(self, txn = None): @@ -144,21 +130,6 @@ class PackageDB(object): # all the list members are removed. self.dr.remove_item(dep_name, repo, txn=txn) - ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn) - search_keys = {} - for (lang, doc) in package_info.summary.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for (lang, doc) in package_info.description.iteritems(): - text = search_keys.get(lang, "") - text += " %s" % doc - search_keys[lang] = text - for lang in search_keys: - text = search_keys[lang] - # FIXME: other languages should be searchable too - if lang in ('en', 'tr'): - pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn) self.d.txn_proc(proc, txn) def remove_repo(self, repo, txn = None): diff --git a/pisi/search/__init__.py b/pisi/search/__init__.py deleted file mode 100644 index cf5a099b..00000000 --- a/pisi/search/__init__.py +++ /dev/null @@ -1,62 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import pisi -import pisi.context as ctx - -class Error(pisi.Error): - pass - -class Exception(pisi.Exception): - pass - -# API - -from invertedindex import InvertedIndex -import preprocess as p - -def init(ids, langs): - "initialize databases" - - assert type(ids)==type([]) - assert type(langs)==type([]) - - ctx.invidx = {} - for id in ids: - ctx.invidx[id] = {} - for lang in langs: - ctx.invidx[id][lang] = InvertedIndex(id, lang) - -def finalize(): - import pisi.context as ctx - - if ctx.invidx: - for id in ctx.invidx.iterkeys(): - for lang in ctx.invidx[id].iterkeys(): - ctx.invidx[id][lang].close() - ctx.invidx = {} - -def add_doc(id, lang, docid, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn) - -def remove_doc(id, lang, docid, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn) - -def query_terms(id, lang, terms, repo = None, txn = None): - terms = p.normalize(lang, terms) - return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn) - -def query(id, lang, str, repo = None, txn = None): - terms = p.preprocess(lang, str) - return query_terms(id, lang, terms, repo = repo, txn = txn) diff --git a/pisi/search/invertedindex.py b/pisi/search/invertedindex.py deleted file mode 100644 index e92aa6ff..00000000 --- a/pisi/search/invertedindex.py +++ /dev/null @@ -1,83 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005 - 2007, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import types - -import pisi.lockeddbshelve as shelve -from pisi.itembyrepodb import ItemByRepoDB -import pisi.itembyrepodb as itembyrepodb - -class InvertedIndex(object): - """a database of term -> set of documents""" - - def __init__(self, id, lang): - self.d = ItemByRepoDB('ii-%s-%s' % (id, lang)) - - def close(self): - self.d.close() - - def has_term(self, term, repo = None, txn = None): - return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn) - - def get_term(self, term, repo = None, txn = None): - """get set of doc ids given term""" - term = shelve.LockedDBShelf.encodekey(term) - def proc(txn): - if not self.has_term(term, repo=repo, txn=txn): - return set() - return self.d.get_item(term, repo=repo, txn=txn) - return self.d.txn_proc(proc, txn) - - def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ): - """get a union of all repository terms, not just the first repo in order. - get only basic repo info from the first repo""" - name = shelve.LockedDBShelf.encodekey(name) - def proc(txn): - terms= set() - if self.d.d.has_key(name): - s = self.d.d.get(name, txn=txn) - for repostr in self.d.order(repo = repo): - if s.has_key(repostr): - terms |= s[repostr] - return terms - return self.d.txn_proc(proc, txn) - - def query(self, terms, repo = None, txn = None): - def proc(txn): - docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ] - if docs: - return reduce(lambda x,y: x.intersection(y), docs) - else: - return set() - return self.d.txn_proc(proc, txn) - - def list_terms(self, repo = None, txn= None): - return self.d.list(f, repo=repo, txn=txn) - - def add_doc(self, doc, terms, repo = None, txn = None): - def f(txn): - for term_i in terms: - term_i = shelve.LockedDBShelf.encodekey(term_i) - term_i_docs = self.get_term(term_i, repo=repo, txn=txn) - term_i_docs.add(doc) - self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update - return self.d.txn_proc(f, txn) - - def remove_doc(self, doc, terms,repo=None, txn=None): - def f(txn): - for term_i in terms: - term_i = shelve.LockedDBShelf.encodekey(term_i) - term_i_docs = self.get_term(term_i,repo=repo, txn=txn) - if doc in term_i_docs: - term_i_docs.remove(doc) - self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update - return self.d.txn_proc(f, txn) diff --git a/pisi/search/preprocess.py b/pisi/search/preprocess.py deleted file mode 100644 index 97964091..00000000 --- a/pisi/search/preprocess.py +++ /dev/null @@ -1,30 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import tokenize -import locale - -def normalize(lang, terms): - if lang == "tr": - old_locale = locale.setlocale(locale.LC_CTYPE) - locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8") - terms = map(lambda x: unicode(x).lower(), terms) - if lang == "tr": - locale.setlocale(locale.LC_CTYPE, old_locale) - unique_terms = set() - for term in terms: - unique_terms.add(unicode(term)) - return list(unique_terms) - -def preprocess(lang, str): - terms = tokenize.tokenize(lang, str) - return normalize(lang, terms) diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py deleted file mode 100644 index 0023e45a..00000000 --- a/pisi/search/tokenize.py +++ /dev/null @@ -1,32 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Copyright (C) 2005-2006, TUBITAK/UEKAE -# -# This program is free software; you can redistribute it and/or modify it under -# the terms of the GNU General Public License as published by the Free -# Software Foundation; either version 2 of the License, or (at your option) -# any later version. -# -# Please read the COPYING file. -# - -import string - -def tokenize(lang, str): - if type(str) != type(unicode()): - str = unicode(str) - sepchars = string.whitespace + string.punctuation - tokens = [] - token = unicode() - for x in str: - if x in sepchars: - if len(token) > 0: - tokens.append(token) - token = unicode() - else: - token += x - - if token: - tokens.append(token) - - return tokens