* fix hyperdandik tokenization a bit.
- include the last token out of the loop if not empty - treat punctuation as whitespace
This commit is contained in:
@@ -21,13 +21,14 @@ def tokenize(lang, str):
|
|||||||
tokens = []
|
tokens = []
|
||||||
token = unicode()
|
token = unicode()
|
||||||
for x in str:
|
for x in str:
|
||||||
if x in string.whitespace:
|
if x in string.whitespace or x in string.punctuation:
|
||||||
if len(token) > 0:
|
if len(token) > 0:
|
||||||
tokens.append(token)
|
tokens.append(token)
|
||||||
token = unicode()
|
token = unicode()
|
||||||
elif x in string.punctuation:
|
|
||||||
pass # eat punctuation
|
|
||||||
else:
|
else:
|
||||||
token += x
|
token += x
|
||||||
|
|
||||||
|
if token:
|
||||||
|
tokens.append(x)
|
||||||
|
|
||||||
return tokens
|
return tokens
|
||||||
|
|||||||
Reference in New Issue
Block a user