Oh, you will love this one...
The DB 101 seems to be, not worked here very well All the search phantasmogoria db code gibbed and replaced by 15 lines of code... + update_repo increased %30-40 percent... + you can now search any keyword _within_ a word... + multi term search speed also increased + decreased code size - nothing really... just some turkish locale lowering stuff in some rare cases which was also fixed by Gurer in the old code not merged but may also be added here some time in the future... oops and special thanks goes to Gurer and Mehmet...
This commit is contained in:
+14
-25
@@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build
|
||||
from pisi.metadata import MetaData
|
||||
from pisi.files import Files
|
||||
from pisi.file import File
|
||||
import pisi.search
|
||||
import pisi.lockeddbshelve as shelve
|
||||
from pisi.version import Version
|
||||
|
||||
@@ -113,7 +112,6 @@ def init(database = True, write = True,
|
||||
ctx.componentdb = pisi.component.ComponentDB()
|
||||
ctx.packagedb = packagedb.init_db()
|
||||
ctx.sourcedb = pisi.sourcedb.init()
|
||||
pisi.search.init(['terms'], ['en', 'tr'])
|
||||
else:
|
||||
ctx.repodb = None
|
||||
ctx.installdb = None
|
||||
@@ -146,7 +144,6 @@ def finalize():
|
||||
if ctx.sourcedb:
|
||||
pisi.sourcedb.finalize()
|
||||
ctx.sourcedb = None
|
||||
pisi.search.finalize()
|
||||
if ctx.dbenv:
|
||||
ctx.dbenv.close()
|
||||
ctx.dbenv_lock.close()
|
||||
@@ -341,30 +338,22 @@ def info_name(package_name, installed=False):
|
||||
files = None
|
||||
return metadata, files
|
||||
|
||||
def search_package_names(query):
|
||||
r = set()
|
||||
packages = ctx.packagedb.list_packages()
|
||||
for pkgname in packages:
|
||||
if query in pkgname:
|
||||
r.add(pkgname)
|
||||
return r
|
||||
def search_package_terms(terms, repo = pisi.itembyrepodb.all):
|
||||
|
||||
def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
r = pisi.search.query_terms('terms', lang, terms, repo = repo)
|
||||
if search_names:
|
||||
for term in terms:
|
||||
r |= search_package_names(term)
|
||||
return r
|
||||
def search(package, term):
|
||||
term = unicode(term).lower()
|
||||
if term in unicode(package.name).lower() or \
|
||||
term in unicode(package.summary).lower() or \
|
||||
term in unicode(package.description).lower():
|
||||
return True
|
||||
|
||||
def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
r = pisi.search.query('terms', lang, query, repo = repo)
|
||||
if search_names:
|
||||
r |= search_package_names(query)
|
||||
return r
|
||||
found = []
|
||||
for name in ctx.packagedb.list_packages(repo):
|
||||
pkg = ctx.packagedb.get_package(name, repo)
|
||||
if terms == filter(lambda x:search(pkg, x), terms):
|
||||
found.append(name)
|
||||
|
||||
return found
|
||||
|
||||
def check(package):
|
||||
md, files = info(package, True)
|
||||
|
||||
@@ -1567,14 +1567,6 @@ in summary, description, and package name fields.
|
||||
default=False, help=_("Show details"))
|
||||
self.parser.add_option_group(group)
|
||||
|
||||
def get_lang(self):
|
||||
lang = ctx.get_option('language')
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
if not lang in ['en', 'tr']:
|
||||
lang = 'en'
|
||||
return lang
|
||||
|
||||
def run(self):
|
||||
|
||||
self.init(database = True, write = False)
|
||||
@@ -1583,7 +1575,7 @@ in summary, description, and package name fields.
|
||||
self.help()
|
||||
return
|
||||
|
||||
r = pisi.api.search_package_terms(self.args, self.get_lang())
|
||||
r = pisi.api.search_package_terms(self.args)
|
||||
ctx.ui.info(_('%s packages found') % len(r))
|
||||
|
||||
ctx.config.options.short = not ctx.config.options.long
|
||||
|
||||
+1
-30
@@ -103,21 +103,7 @@ class PackageDB(object):
|
||||
self.dr.add_item(dep_name, [ (name, dep) ], repo, txn)
|
||||
# add component
|
||||
ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn)
|
||||
# index summary and description
|
||||
search_keys = {}
|
||||
for (lang, doc) in package_info.summary.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for (lang, doc) in package_info.description.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for lang in search_keys:
|
||||
text = search_keys[lang]
|
||||
# FIXME: other languages should be searchable too
|
||||
if lang in ('en', 'tr'):
|
||||
pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
|
||||
|
||||
ctx.txn_proc(proc, txn)
|
||||
|
||||
def clear(self, txn = None):
|
||||
@@ -144,21 +130,6 @@ class PackageDB(object):
|
||||
# all the list members are removed.
|
||||
self.dr.remove_item(dep_name, repo, txn=txn)
|
||||
|
||||
ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn)
|
||||
search_keys = {}
|
||||
for (lang, doc) in package_info.summary.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for (lang, doc) in package_info.description.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for lang in search_keys:
|
||||
text = search_keys[lang]
|
||||
# FIXME: other languages should be searchable too
|
||||
if lang in ('en', 'tr'):
|
||||
pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
|
||||
self.d.txn_proc(proc, txn)
|
||||
|
||||
def remove_repo(self, repo, txn = None):
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import pisi
|
||||
import pisi.context as ctx
|
||||
|
||||
class Error(pisi.Error):
|
||||
pass
|
||||
|
||||
class Exception(pisi.Exception):
|
||||
pass
|
||||
|
||||
# API
|
||||
|
||||
from invertedindex import InvertedIndex
|
||||
import preprocess as p
|
||||
|
||||
def init(ids, langs):
|
||||
"initialize databases"
|
||||
|
||||
assert type(ids)==type([])
|
||||
assert type(langs)==type([])
|
||||
|
||||
ctx.invidx = {}
|
||||
for id in ids:
|
||||
ctx.invidx[id] = {}
|
||||
for lang in langs:
|
||||
ctx.invidx[id][lang] = InvertedIndex(id, lang)
|
||||
|
||||
def finalize():
|
||||
import pisi.context as ctx
|
||||
|
||||
if ctx.invidx:
|
||||
for id in ctx.invidx.iterkeys():
|
||||
for lang in ctx.invidx[id].iterkeys():
|
||||
ctx.invidx[id][lang].close()
|
||||
ctx.invidx = {}
|
||||
|
||||
def add_doc(id, lang, docid, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn)
|
||||
|
||||
def remove_doc(id, lang, docid, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn)
|
||||
|
||||
def query_terms(id, lang, terms, repo = None, txn = None):
|
||||
terms = p.normalize(lang, terms)
|
||||
return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn)
|
||||
|
||||
def query(id, lang, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
return query_terms(id, lang, terms, repo = repo, txn = txn)
|
||||
@@ -1,83 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import types
|
||||
|
||||
import pisi.lockeddbshelve as shelve
|
||||
from pisi.itembyrepodb import ItemByRepoDB
|
||||
import pisi.itembyrepodb as itembyrepodb
|
||||
|
||||
class InvertedIndex(object):
|
||||
"""a database of term -> set of documents"""
|
||||
|
||||
def __init__(self, id, lang):
|
||||
self.d = ItemByRepoDB('ii-%s-%s' % (id, lang))
|
||||
|
||||
def close(self):
|
||||
self.d.close()
|
||||
|
||||
def has_term(self, term, repo = None, txn = None):
|
||||
return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn)
|
||||
|
||||
def get_term(self, term, repo = None, txn = None):
|
||||
"""get set of doc ids given term"""
|
||||
term = shelve.LockedDBShelf.encodekey(term)
|
||||
def proc(txn):
|
||||
if not self.has_term(term, repo=repo, txn=txn):
|
||||
return set()
|
||||
return self.d.get_item(term, repo=repo, txn=txn)
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ):
|
||||
"""get a union of all repository terms, not just the first repo in order.
|
||||
get only basic repo info from the first repo"""
|
||||
name = shelve.LockedDBShelf.encodekey(name)
|
||||
def proc(txn):
|
||||
terms= set()
|
||||
if self.d.d.has_key(name):
|
||||
s = self.d.d.get(name, txn=txn)
|
||||
for repostr in self.d.order(repo = repo):
|
||||
if s.has_key(repostr):
|
||||
terms |= s[repostr]
|
||||
return terms
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def query(self, terms, repo = None, txn = None):
|
||||
def proc(txn):
|
||||
docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ]
|
||||
if docs:
|
||||
return reduce(lambda x,y: x.intersection(y), docs)
|
||||
else:
|
||||
return set()
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def list_terms(self, repo = None, txn= None):
|
||||
return self.d.list(f, repo=repo, txn=txn)
|
||||
|
||||
def add_doc(self, doc, terms, repo = None, txn = None):
|
||||
def f(txn):
|
||||
for term_i in terms:
|
||||
term_i = shelve.LockedDBShelf.encodekey(term_i)
|
||||
term_i_docs = self.get_term(term_i, repo=repo, txn=txn)
|
||||
term_i_docs.add(doc)
|
||||
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
|
||||
return self.d.txn_proc(f, txn)
|
||||
|
||||
def remove_doc(self, doc, terms,repo=None, txn=None):
|
||||
def f(txn):
|
||||
for term_i in terms:
|
||||
term_i = shelve.LockedDBShelf.encodekey(term_i)
|
||||
term_i_docs = self.get_term(term_i,repo=repo, txn=txn)
|
||||
if doc in term_i_docs:
|
||||
term_i_docs.remove(doc)
|
||||
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
|
||||
return self.d.txn_proc(f, txn)
|
||||
@@ -1,30 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import tokenize
|
||||
import locale
|
||||
|
||||
def normalize(lang, terms):
|
||||
if lang == "tr":
|
||||
old_locale = locale.setlocale(locale.LC_CTYPE)
|
||||
locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8")
|
||||
terms = map(lambda x: unicode(x).lower(), terms)
|
||||
if lang == "tr":
|
||||
locale.setlocale(locale.LC_CTYPE, old_locale)
|
||||
unique_terms = set()
|
||||
for term in terms:
|
||||
unique_terms.add(unicode(term))
|
||||
return list(unique_terms)
|
||||
|
||||
def preprocess(lang, str):
|
||||
terms = tokenize.tokenize(lang, str)
|
||||
return normalize(lang, terms)
|
||||
@@ -1,32 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import string
|
||||
|
||||
def tokenize(lang, str):
|
||||
if type(str) != type(unicode()):
|
||||
str = unicode(str)
|
||||
sepchars = string.whitespace + string.punctuation
|
||||
tokens = []
|
||||
token = unicode()
|
||||
for x in str:
|
||||
if x in sepchars:
|
||||
if len(token) > 0:
|
||||
tokens.append(token)
|
||||
token = unicode()
|
||||
else:
|
||||
token += x
|
||||
|
||||
if token:
|
||||
tokens.append(token)
|
||||
|
||||
return tokens
|
||||
Reference in New Issue
Block a user