From 81d11c81725920ac3826d22f0d4d0719f01117e7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eray=20=C3=96zkural?= Date: Mon, 12 Dec 2005 23:40:59 +0000 Subject: [PATCH] * fix hyperdandik tokenization a bit. - include the last token out of the loop if not empty - treat punctuation as whitespace --- pisi/search/tokenize.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/pisi/search/tokenize.py b/pisi/search/tokenize.py index 4d767127..e479ca86 100644 --- a/pisi/search/tokenize.py +++ b/pisi/search/tokenize.py @@ -21,13 +21,14 @@ def tokenize(lang, str): tokens = [] token = unicode() for x in str: - if x in string.whitespace: + if x in string.whitespace or x in string.punctuation: if len(token) > 0: tokens.append(token) token = unicode() - elif x in string.punctuation: - pass # eat punctuation else: token += x + + if token: + tokens.append(x) return tokens