From 56084de2a5fce1a9f2447876e3acda0453c853d0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?G=C3=BCrer=20=C3=96zen?= Date: Fri, 3 Nov 2006 22:43:02 +0000 Subject: [PATCH] =?UTF-8?q?preprocess=20fonksiyonu=20art=C4=B1k=20do=C4=9F?= =?UTF-8?q?ru=20=C3=A7al=C4=B1=C5=9F=C4=B1yor.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit rebuild-db de 1 snlik falan bir hızlanma var, pek kayda değmedi o açıdan, asıl hız kazancı için db değişikliği gerekecek :( pisi search çok daha iyi çalışıyor, ama gene de bazı aksamalar var, sanırım api.py yada InvertedIndex içindeki kısımlarda da bir sorun var. --- pisi/search/__init__.py | 4 ++-- pisi/search/preprocess.py | 35 +++++++++++------------------------ pisi/search/tokenize.py | 5 +++-- 3 files changed, 16 insertions(+), 28 deletions(-) diff --git a/pisi/search/__init__.py b/pisi/search/__init__.py index 9cb0fc48..cf5a099b 100644 --- a/pisi/search/__init__.py +++ b/pisi/search/__init__.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- # -# Copyright (C) 2005, TUBITAK/UEKAE +# Copyright (C) 2005-2006, TUBITAK/UEKAE # # This program is free software; you can redistribute it and/or modify it under # the terms of the GNU General Public License as published by the Free @@ -54,7 +54,7 @@ def remove_doc(id, lang, docid, str, repo = None, txn = None): ctx.invidx[id][lang].remove_doc(docid, terms, repo = repo, txn = txn) def query_terms(id, lang, terms, repo = None, txn = None): - terms = map(lambda x: p.lower(lang, x), terms) + terms = p.normalize(lang, terms) return ctx.invidx[id][lang].query(terms, repo = repo, txn = txn) def query(id, lang, str, repo = None, txn = None): diff --git a/pisi/search/preprocess.py b/pisi/search/preprocess.py index 14df5d47..b3790325 100644 --- a/pisi/search/preprocess.py +++ b/pisi/search/preprocess.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- # -# Copyright (C) 2005, TUBITAK/UEKAE +# Copyright (C) 2005-2006, TUBITAK/UEKAE # # This program is free software; you can redistribute it and/or modify it under # the terms of the GNU General Public License as published by the Free @@ -11,30 +11,17 @@ # import tokenize +import locale -def lowly_python(str): - def lowly_char(c): - if c=='I': - lowly = 'i' # because of some fools we can't choose locale in lower - else: - lowly = c.lower() - return c - - r = "" - for c in str: - r += lowly_char(c) - return r - -def lower(lang, str): - if lang=='tr': - return lowly_python(str) - else: - return str.lower() +def normalize(lang, terms): + if lang == "tr": + old_locale = locale.setlocale(locale.LC_CTYPE) + locale.setlocale(locale.LC_CTYPE, "tr_TR.UTF-8") + terms = map(lambda x: unicode(x).lower(), terms) + if lang == "tr": + locale.setlocale(locale.LC_CTYPE, old_locale) + return terms def preprocess(lang, str): terms = tokenize.tokenize(lang, str) - - # normalize - terms = map(lambda x: lower(lang, x), terms) - - return terms + return normalize(lang, terms) diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py index 51d4dce1..0023e45a 100644 --- a/pisi/search/tokenize.py +++ b/pisi/search/tokenize.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- # -# Copyright (C) 2005, TUBITAK/UEKAE +# Copyright (C) 2005-2006, TUBITAK/UEKAE # # This program is free software; you can redistribute it and/or modify it under # the terms of the GNU General Public License as published by the Free @@ -15,10 +15,11 @@ import string def tokenize(lang, str): if type(str) != type(unicode()): str = unicode(str) + sepchars = string.whitespace + string.punctuation tokens = [] token = unicode() for x in str: - if x in string.whitespace or x in string.punctuation: + if x in sepchars: if len(token) > 0: tokens.append(token) token = unicode()