diff --git a/pisi.e3p b/pisi.e3p
index a5160038..883f46bd 100644
--- a/pisi.e3p
+++ b/pisi.e3p
@@ -1,7 +1,7 @@
-
+
Python
@@ -438,6 +438,26 @@
pisi
file.py
+
+ pisi
+ search
+ invertedindex.py
+
+
+ pisi
+ search
+ tokenize.py
+
+
+ pisi
+ search
+ __init__.py
+
+
+ pisi
+ search
+ preprocess.py
+
@@ -480,9 +500,9 @@
-
-
-
+
+
+
diff --git a/pisi/lockeddbshelve.py b/pisi/lockeddbshelve.py
index dafb7e56..ff2534ce 100644
--- a/pisi/lockeddbshelve.py
+++ b/pisi/lockeddbshelve.py
@@ -16,6 +16,7 @@ import bsddb.dbshelve as shelve
import bsddb.db as db
import os
import fcntl
+import types
import gettext
__trans = gettext.translation('pisi', fallback=True)
@@ -77,7 +78,6 @@ class LockedDBShelf(shelve.DBShelf):
except IOError:
raise Error(_("Another instance of PISI is running. Try later!"))
-
def close(self):
if self.closed:
return
@@ -89,3 +89,13 @@ class LockedDBShelf(shelve.DBShelf):
def unlock(self):
self.lockfile.close()
os.unlink(self.filename + '.lock')
+
+ @staticmethod
+ def encodekey(key):
+ '''utility method for dbs that must store unicodes in keys'''
+ if type(key)==types.UnicodeType:
+ return key.encode('utf-8')
+ elif type(key)==types.StringType:
+ return key
+ else:
+ raise Error('Key must be either string or unicode')
diff --git a/pisi/search/__init__.py b/pisi/search/__init__.py
index ea52d49a..3af54fac 100644
--- a/pisi/search/__init__.py
+++ b/pisi/search/__init__.py
@@ -13,6 +13,7 @@
# Author: Eray Ozkural
import pisi
+import pisi.context as ctx
class Error(pisi.Error):
pass
@@ -20,13 +21,16 @@ class Error(pisi.Error):
class Exception(pisi.Exception):
pass
-# API
+# API
from invertedindex import InvertedIndex
+from preprocess import preprocess
def init(ids, langs):
"initialize databases"
- import pisi.context as ctx
+
+ assert type(ids)==type([])
+ assert type(langs)==type([])
ctx.invidx = {}
for id in ids:
@@ -44,10 +48,11 @@ def finalize():
ctx.invidx = {}
def add_doc(id, lang, docid, str):
- pass
+ terms = preprocess(lang, str)
+ ctx.invidx[id][lang].add_doc(docid, terms)
def remove_doc(id, lang, docid, str):
- pass
+ ctx.invidx[id][lang].remove_doc(docid)
-def query_terms(id, lang, terms):
- pass
+def query(id, lang, terms):
+ return ctx.invidx[id][lang].query(terms)
diff --git a/pisi/search/invertedindex.py b/pisi/search/invertedindex.py
index d05a54ae..25e2af34 100644
--- a/pisi/search/invertedindex.py
+++ b/pisi/search/invertedindex.py
@@ -11,6 +11,11 @@
#
# Author: Eray Ozkural
+import types
+
+import pisi.lockeddbshelve as shelve
+
+
class InvertedIndex(object):
"""a database of term -> set of documents"""
@@ -21,14 +26,19 @@ class InvertedIndex(object):
self.d.close()
def has_term(self, term):
- return self.d.has_key(str(term))
+ return self.d.has_key(shelve.LockedDBShelf.encodekey(term))
def get_term(self, term):
- term = str(term)
+ """get set of doc ids given term"""
+ term = shelve.LockedDBShelf.encodekey(term)
if not self.has_term(term):
self.d[term] = set()
return self.d[term]
+ def query(self, terms):
+ docs = [ self.get_term(x) for x in terms ]
+ return reduce(lambda x,y: x.union(y), docs)
+
def list_terms(self):
list = []
for term in self.d.iterkeys():
@@ -37,12 +47,14 @@ class InvertedIndex(object):
def add_doc(self, doc, terms):
for term_i in terms:
+ term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i)
term_i_docs.add(doc)
self.d[term_i] = term_i_docs # update
def remove_doc(self, doc, terms):
for term_i in terms:
+ term_i = shelve.LockedDBShelf.encodekey(term_i)
term_i_docs = self.get_term(term_i)
term_i_docs.remove(doc)
self.d[term_i] = term_i_docs # update
diff --git a/pisi/search/preprocess.py b/pisi/search/preprocess.py
index 60712c02..aeb97479 100644
--- a/pisi/search/preprocess.py
+++ b/pisi/search/preprocess.py
@@ -1,3 +1,16 @@
+# -*- coding: utf-8 -*-
+#
+# Copyright (C) 2005, TUBITAK/UEKAE
+#
+# This program is free software; you can redistribute it and/or modify it under
+# the terms of the GNU General Public License as published by the Free
+# Software Foundation; either version 2 of the License, or (at your option)
+# any later version.
+#
+# Please read the COPYING file.
+#
-def preprocess(str, lang):
- if
+import tokenize
+
+def preprocess(lang, str):
+ return tokenize.tokenize(lang, str)
diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py
index 3c940a43..87cbdc06 100644
--- a/pisi/search/tokenize.py
+++ b/pisi/search/tokenize.py
@@ -11,7 +11,21 @@
#
# Author: Eray Ozkural
#
-# rev 1: very little tokenization, for testing
+# rev 1: very simple tokenization, for testing
-def tokenize(lang, string):
- pass
+import string
+
+def tokenize(lang, str):
+ if type(str) != type(unicode()):
+ str = unicode(str)
+ tokens = []
+ token = unicode()
+ for x in str:
+ if x in string.whitespace:
+ if len(token) > 0:
+ tokens.append(token)
+ token = unicode()
+ else:
+ token += x
+
+ return tokens
diff --git a/tests/searchtests.py b/tests/searchtests.py
new file mode 100644
index 00000000..98781a1e
--- /dev/null
+++ b/tests/searchtests.py
@@ -0,0 +1,39 @@
+# -*- coding: utf-8 -*-
+#
+# Copyright (C) 2005, TUBITAK/UEKAE
+#
+# This program is free software; you can redistribute it and/or modify it under
+# the terms of the GNU General Public License as published by the Free
+# Software Foundation; either version 2 of the License, or (at your option)
+# any later version.
+#
+# Please read the COPYING file.
+#
+
+import unittest
+import os
+
+import pisi.search
+
+import testcase
+
+class SearchTestCase(testcase.TestCase):
+
+ def setUp(self):
+ testcase.TestCase.setUp(self, database = False)
+
+ def testSearch(self):
+ doc1 = "A set object is an unordered collection of immutable values."
+ doc2 = "Being an unordered collection, sets do not record element position or order of insertion."
+ doc3 = "There are currently two builtin set types, set and frozenset"
+ pisi.search.init(['test'], ['en'])
+ pisi.search.add_doc('test', 'en', 1, doc1)
+ pisi.search.add_doc('test', 'en', 2, doc2)
+ pisi.search.add_doc('test', 'en', 3, doc3)
+ q1 = pisi.search.query('test', 'en', ['set'])
+ self.assertEqual(q1, set([1,3]))
+ q2 = pisi.search.query('test', 'en', ['an', 'collection'])
+ self.assertEqual(q2, set([1,2]))
+ pisi.search.finalize()
+
+suite = unittest.makeSuite(SearchTestCase)