summaryrefslogtreecommitdiff
path: root/corpus
diff options
context:
space:
mode:
authorChris Dyer <cdyer@allegro.clab.cs.cmu.edu>2014-03-18 02:05:25 -0400
committerChris Dyer <cdyer@allegro.clab.cs.cmu.edu>2014-03-18 02:05:25 -0400
commit2a9ee1febae6a63173f74ae24e2bfe439e409525 (patch)
tree44b930ed4bdaedc4e6c6f43af18474fb367b97fd /corpus
parent55beb71dbbfe8421a8e8be9d2d7868a5d87d77d5 (diff)
chris edits
Diffstat (limited to 'corpus')
-rwxr-xr-xcorpus/support/tokenizer.pl4
1 files changed, 4 insertions, 0 deletions
diff --git a/corpus/support/tokenizer.pl b/corpus/support/tokenizer.pl
index 7771201f..f57bc87a 100755
--- a/corpus/support/tokenizer.pl
+++ b/corpus/support/tokenizer.pl
@@ -240,6 +240,10 @@ sub proc_token {
return $token;
}
+ if($token =~ /^\d+(.\d+)+(亿|百万|万|千)?$/){
+ return $token;
+ }
+
## 1,234,345.34
if($token =~ /^\d+(\.\d{3})*,\d+$/){
## number