diff --git a/pisi/search/__init__.py b/pisi/search/__init__.py new file mode 100644 index 00000000..ea52d49a --- /dev/null +++ b/pisi/search/__init__.py @@ -0,0 +1,53 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2005, TUBITAK/UEKAE +# +# This program is free software; you can redistribute it and/or modify it under +# the terms of the GNU General Public License as published by the Free +# Software Foundation; either version 2 of the License, or (at your option) +# any later version. +# +# Please read the COPYING file. +# +# +# Author: Eray Ozkural + +import pisi + +class Error(pisi.Error): + pass + +class Exception(pisi.Exception): + pass + +# API + +from invertedindex import InvertedIndex + +def init(ids, langs): + "initialize databases" + import pisi.context as ctx + + ctx.invidx = {} + for id in ids: + ctx.invidx[id] = {} + for lang in langs: + ctx.invidx[id][lang] = InvertedIndex(id, lang) + +def finalize(): + import pisi.context as ctx + + for id in ctx.invidx.iterkeys(): + for lang in ctx.invidx[id].iterkeys(): + ctx.invidx[id][lang].close() + + ctx.invidx = {} + +def add_doc(id, lang, docid, str): + pass + +def remove_doc(id, lang, docid, str): + pass + +def query_terms(id, lang, terms): + pass diff --git a/pisi/search/invertedindex.py b/pisi/search/invertedindex.py new file mode 100644 index 00000000..d05a54ae --- /dev/null +++ b/pisi/search/invertedindex.py @@ -0,0 +1,48 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2005, TUBITAK/UEKAE +# +# This program is free software; you can redistribute it and/or modify it under +# the terms of the GNU General Public License as published by the Free +# Software Foundation; either version 2 of the License, or (at your option) +# any later version. +# +# Please read the COPYING file. +# +# Author: Eray Ozkural + +class InvertedIndex(object): + """a database of term -> set of documents""" + + def __init__(self, id, lang): + self.d = shelve.LockedDBShelf('ii-%s-%s' % (id, lang)) + + def close(self): + self.d.close() + + def has_term(self, term): + return self.d.has_key(str(term)) + + def get_term(self, term): + term = str(term) + if not self.has_term(term): + self.d[term] = set() + return self.d[term] + + def list_terms(self): + list = [] + for term in self.d.iterkeys(): + list.append(term) + return list + + def add_doc(self, doc, terms): + for term_i in terms: + term_i_docs = self.get_term(term_i) + term_i_docs.add(doc) + self.d[term_i] = term_i_docs # update + + def remove_doc(self, doc, terms): + for term_i in terms: + term_i_docs = self.get_term(term_i) + term_i_docs.remove(doc) + self.d[term_i] = term_i_docs # update diff --git a/pisi/search/preprocess.py b/pisi/search/preprocess.py new file mode 100644 index 00000000..60712c02 --- /dev/null +++ b/pisi/search/preprocess.py @@ -0,0 +1,3 @@ + +def preprocess(str, lang): + if diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py new file mode 100644 index 00000000..3c940a43 --- /dev/null +++ b/pisi/search/tokenize.py @@ -0,0 +1,17 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2005, TUBITAK/UEKAE +# +# This program is free software; you can redistribute it and/or modify it under +# the terms of the GNU General Public License as published by the Free +# Software Foundation; either version 2 of the License, or (at your option) +# any later version. +# +# Please read the COPYING file. +# +# Author: Eray Ozkural +# +# rev 1: very little tokenization, for testing + +def tokenize(lang, string): + pass