* fix hyperdandik tokenization a bit.

- include the last token out of the loop if not empty
  - treat punctuation as whitespace
This commit is contained in:
Eray Özkural
2005-12-12 23:40:59 +00:00
parent cb7893610e
commit 81d11c8172
+4 -3
View File
@@ -21,13 +21,14 @@ def tokenize(lang, str):
tokens = [] tokens = []
token = unicode() token = unicode()
for x in str: for x in str:
if x in string.whitespace: if x in string.whitespace or x in string.punctuation:
if len(token) > 0: if len(token) > 0:
tokens.append(token) tokens.append(token)
token = unicode() token = unicode()
elif x in string.punctuation:
pass # eat punctuation
else: else:
token += x token += x
if token:
tokens.append(x)
return tokens return tokens