* merge search db and code removal from trunk..
- update_db operation is now %40+ faster... - also search operations are faster...
This commit is contained in:
+15
-26
@@ -1,6 +1,6 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005, TUBITAK/UEKAE
|
||||
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
@@ -43,7 +43,6 @@ from pisi.atomicoperations import resurrect_package, build
|
||||
from pisi.metadata import MetaData
|
||||
from pisi.files import Files
|
||||
from pisi.file import File
|
||||
import pisi.search
|
||||
import pisi.lockeddbshelve as shelve
|
||||
from pisi.version import Version
|
||||
|
||||
@@ -113,7 +112,6 @@ def init(database = True, write = True,
|
||||
ctx.componentdb = pisi.component.ComponentDB()
|
||||
ctx.packagedb = packagedb.init_db()
|
||||
ctx.sourcedb = pisi.sourcedb.init()
|
||||
pisi.search.init(['terms'], ['en', 'tr'])
|
||||
else:
|
||||
ctx.repodb = None
|
||||
ctx.installdb = None
|
||||
@@ -146,7 +144,6 @@ def finalize():
|
||||
if ctx.sourcedb:
|
||||
pisi.sourcedb.finalize()
|
||||
ctx.sourcedb = None
|
||||
pisi.search.finalize()
|
||||
if ctx.dbenv:
|
||||
ctx.dbenv.close()
|
||||
ctx.dbenv_lock.close()
|
||||
@@ -341,30 +338,22 @@ def info_name(package_name, installed=False):
|
||||
files = None
|
||||
return metadata, files
|
||||
|
||||
def search_package_names(query):
|
||||
r = set()
|
||||
packages = ctx.packagedb.list_packages()
|
||||
for pkgname in packages:
|
||||
if query in pkgname:
|
||||
r.add(pkgname)
|
||||
return r
|
||||
def search_package_terms(terms, repo = pisi.itembyrepodb.all):
|
||||
|
||||
def search_package_terms(terms, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
r = pisi.search.query_terms('terms', lang, terms, repo = repo)
|
||||
if search_names:
|
||||
for term in terms:
|
||||
r |= search_package_names(term)
|
||||
return r
|
||||
def search(package, term):
|
||||
term = unicode(term).lower()
|
||||
if term in unicode(package.name).lower() or \
|
||||
term in unicode(package.summary).lower() or \
|
||||
term in unicode(package.description).lower():
|
||||
return True
|
||||
|
||||
def search_package(query, lang = None, search_names = True, repo = pisi.itembyrepodb.all):
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
r = pisi.search.query('terms', lang, query, repo = repo)
|
||||
if search_names:
|
||||
r |= search_package_names(query)
|
||||
return r
|
||||
found = []
|
||||
for name in ctx.packagedb.list_packages(repo):
|
||||
pkg = ctx.packagedb.get_package(name, repo)
|
||||
if terms == filter(lambda x:search(pkg, x), terms):
|
||||
found.append(name)
|
||||
|
||||
return found
|
||||
|
||||
def check(package):
|
||||
md, files = info(package, True)
|
||||
|
||||
@@ -1510,14 +1510,6 @@ in summary, description, and package name fields.
|
||||
default=False, help=_("Show details"))
|
||||
self.parser.add_option_group(group)
|
||||
|
||||
def get_lang(self):
|
||||
lang = ctx.get_option('language')
|
||||
if not lang:
|
||||
lang = pisi.pxml.autoxml.LocalText.get_lang()
|
||||
if not lang in ['en', 'tr']:
|
||||
lang = 'en'
|
||||
return lang
|
||||
|
||||
def run(self):
|
||||
|
||||
self.init(database = True, write = False)
|
||||
@@ -1526,7 +1518,7 @@ in summary, description, and package name fields.
|
||||
self.help()
|
||||
return
|
||||
|
||||
r = pisi.api.search_package_terms(self.args, self.get_lang())
|
||||
r = pisi.api.search_package_terms(self.args)
|
||||
ctx.ui.info(_('%s packages found') % len(r))
|
||||
|
||||
ctx.config.options.short = not ctx.config.options.long
|
||||
|
||||
+4
-30
@@ -1,6 +1,6 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005, TUBITAK/UEKAE
|
||||
# Copyright (C) 2005 - 2007, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
@@ -103,21 +103,7 @@ class PackageDB(object):
|
||||
self.dr.add_item(dep_name, [ (name, dep) ], repo, txn)
|
||||
# add component
|
||||
ctx.componentdb.add_package(package_info.partOf, package_info.name, repo, txn)
|
||||
# index summary and description
|
||||
search_keys = {}
|
||||
for (lang, doc) in package_info.summary.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for (lang, doc) in package_info.description.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for lang in search_keys:
|
||||
text = search_keys[lang]
|
||||
# FIXME: other languages should be searchable too
|
||||
if lang in ('en', 'tr'):
|
||||
pisi.search.add_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
|
||||
|
||||
ctx.txn_proc(proc, txn)
|
||||
|
||||
def clear(self, txn = None):
|
||||
@@ -144,21 +130,9 @@ class PackageDB(object):
|
||||
# all the list members are removed.
|
||||
self.dr.remove_item(dep_name, repo, txn=txn)
|
||||
|
||||
# remove from component
|
||||
ctx.componentdb.remove_package(package_info.partOf, package_info.name, repo, txn)
|
||||
search_keys = {}
|
||||
for (lang, doc) in package_info.summary.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for (lang, doc) in package_info.description.iteritems():
|
||||
text = search_keys.get(lang, "")
|
||||
text += " %s" % doc
|
||||
search_keys[lang] = text
|
||||
for lang in search_keys:
|
||||
text = search_keys[lang]
|
||||
# FIXME: other languages should be searchable too
|
||||
if lang in ('en', 'tr'):
|
||||
pisi.search.remove_doc('terms', lang, package_info.name, doc, repo=repo, txn=txn)
|
||||
|
||||
self.d.txn_proc(proc, txn)
|
||||
|
||||
def remove_repo(self, repo, txn = None):
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import pisi
|
||||
import pisi.context as ctx
|
||||
|
||||
class Error(pisi.Error):
|
||||
pass
|
||||
|
||||
class Exception(pisi.Exception):
|
||||
pass
|
||||
|
||||
# API
|
||||
|
||||
from invertedindex import InvertedIndex
|
||||
import preprocess as p
|
||||
|
||||
def init(ids, langs):
|
||||
"initialize databases"
|
||||
|
||||
assert type(ids)==type([])
|
||||
assert type(langs)==type([])
|
||||
|
||||
ctx.invidx = {}
|
||||
for id in ids:
|
||||
ctx.invidx[id] = {}
|
||||
for lang in langs:
|
||||
ctx.invidx[id][lang] = InvertedIndex(id, lang)
|
||||
|
||||
def finalize():
|
||||
import pisi.context as ctx
|
||||
|
||||
if ctx.invidx:
|
||||
for id in ctx.invidx.iterkeys():
|
||||
for lang in ctx.invidx[id].iterkeys():
|
||||
ctx.invidx[id][lang].close()
|
||||
ctx.invidx = {}
|
||||
|
||||
def add_doc(id, lang, docid, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
ctx.invidx[id][lang].add_doc(docid, terms, repo=repo, txn=txn)
|
||||
|
||||
def remove_doc(id, lang, docid, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn)
|
||||
|
||||
def query_terms(id, lang, terms, repo = None, txn = None):
|
||||
terms = p.normalize(lang, terms)
|
||||
return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn)
|
||||
|
||||
def query(id, lang, str, repo = None, txn = None):
|
||||
terms = p.preprocess(lang, str)
|
||||
return query_terms(id, lang, terms, repo = repo, txn = txn)
|
||||
@@ -1,83 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import types
|
||||
|
||||
import pisi.lockeddbshelve as shelve
|
||||
from pisi.itembyrepodb import ItemByRepoDB
|
||||
import pisi.itembyrepodb as itembyrepodb
|
||||
|
||||
class InvertedIndex(object):
|
||||
"""a database of term -> set of documents"""
|
||||
|
||||
def __init__(self, id, lang):
|
||||
self.d = ItemByRepoDB('ii-%s-%s' % (id, lang))
|
||||
|
||||
def close(self):
|
||||
self.d.close()
|
||||
|
||||
def has_term(self, term, repo = None, txn = None):
|
||||
return self.d.has_key(shelve.LockedDBShelf.encodekey(term), repo=repo,txn=txn)
|
||||
|
||||
def get_term(self, term, repo = None, txn = None):
|
||||
"""get set of doc ids given term"""
|
||||
term = shelve.LockedDBShelf.encodekey(term)
|
||||
def proc(txn):
|
||||
if not self.has_term(term, repo=repo, txn=txn):
|
||||
return set()
|
||||
return self.d.get_item(term, repo=repo, txn=txn)
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def get_union_term(self, name, txn = None, repo = itembyrepodb.repos ):
|
||||
"""get a union of all repository terms, not just the first repo in order.
|
||||
get only basic repo info from the first repo"""
|
||||
name = shelve.LockedDBShelf.encodekey(name)
|
||||
def proc(txn):
|
||||
terms= set()
|
||||
if self.d.d.has_key(name):
|
||||
s = self.d.d.get(name, txn=txn)
|
||||
for repostr in self.d.order(repo = repo):
|
||||
if s.has_key(repostr):
|
||||
terms |= s[repostr]
|
||||
return terms
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def query(self, terms, repo = None, txn = None):
|
||||
def proc(txn):
|
||||
docs = [ self.get_union_term(x, repo=repo, txn=txn) for x in terms ]
|
||||
if docs:
|
||||
return reduce(lambda x,y: x.intersection(y), docs)
|
||||
else:
|
||||
return set()
|
||||
return self.d.txn_proc(proc, txn)
|
||||
|
||||
def list_terms(self, repo = None, txn= None):
|
||||
return self.d.list(f, repo=repo, txn=txn)
|
||||
|
||||
def add_doc(self, doc, terms, repo = None, txn = None):
|
||||
def f(txn):
|
||||
for term_i in terms:
|
||||
term_i = shelve.LockedDBShelf.encodekey(term_i)
|
||||
term_i_docs = self.get_term(term_i, repo=repo, txn=txn)
|
||||
term_i_docs.add(doc)
|
||||
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
|
||||
return self.d.txn_proc(f, txn)
|
||||
|
||||
def remove_doc(self, doc, terms,repo=None, txn=None):
|
||||
def f(txn):
|
||||
for term_i in terms:
|
||||
term_i = shelve.LockedDBShelf.encodekey(term_i)
|
||||
term_i_docs = self.get_term(term_i,repo=repo, txn=txn)
|
||||
if doc in term_i_docs:
|
||||
term_i_docs.remove(doc)
|
||||
self.d.add_item(term_i, term_i_docs, repo=repo, txn=txn) # update
|
||||
return self.d.txn_proc(f, txn)
|
||||
@@ -1,30 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import tokenize
|
||||
import locale
|
||||
|
||||
def normalize(lang, terms):
|
||||
if lang == "tr":
|
||||
old_locale = locale.setlocale(locale.LC_CTYPE)
|
||||
locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8")
|
||||
terms = map(lambda x: unicode(x).lower(), terms)
|
||||
if lang == "tr":
|
||||
locale.setlocale(locale.LC_CTYPE, old_locale)
|
||||
unique_terms = set()
|
||||
for term in terms:
|
||||
unique_terms.add(unicode(term))
|
||||
return list(unique_terms)
|
||||
|
||||
def preprocess(lang, str):
|
||||
terms = tokenize.tokenize(lang, str)
|
||||
return normalize(lang, terms)
|
||||
@@ -1,32 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Copyright (C) 2005-2006, TUBITAK/UEKAE
|
||||
#
|
||||
# This program is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by the Free
|
||||
# Software Foundation; either version 2 of the License, or (at your option)
|
||||
# any later version.
|
||||
#
|
||||
# Please read the COPYING file.
|
||||
#
|
||||
|
||||
import string
|
||||
|
||||
def tokenize(lang, str):
|
||||
if type(str) != type(unicode()):
|
||||
str = unicode(str)
|
||||
sepchars = string.whitespace + string.punctuation
|
||||
tokens = []
|
||||
token = unicode()
|
||||
for x in str:
|
||||
if x in sepchars:
|
||||
if len(token) > 0:
|
||||
tokens.append(token)
|
||||
token = unicode()
|
||||
else:
|
||||
token += x
|
||||
|
||||
if token:
|
||||
tokens.append(token)
|
||||
|
||||
return tokens
|
||||
Reference in New Issue
Block a user