Merge branch 'master' of https://github.com/redpony/cdec

author: Chris Dyer <cdyer@cs.cmu.edu> 2012-10-11 14:06:32 -0400
committer: Chris Dyer <cdyer@cs.cmu.edu> 2012-10-11 14:06:32 -0400
commit: 07ea7b64b6f85e5798a8068453ed9fd2b97396db (patch)
tree: 644496a1690d84d82a396bbc1e39160788beb2cd /gi/morf-segmentation/vocabextractor.sh
parent: 37b9e45e5cb29d708f7249dbe0b0fb27685282a0 (diff)
parent: a36fcc5d55c1de84ae68c1091ebff2b1c32dc3b7 (diff)
1 files changed, 0 insertions, 40 deletions
diff --git a/gi/morf-segmentation/vocabextractor.sh b/gi/morf-segmentation/vocabextractor.sh
deleted file mode 100755
index 00ae7109..00000000
--- a/gi/morf-segmentation/vocabextractor.sh
+++ /dev/null
@@ -1,40 +0,0 @@
-#!/bin/bash
-
-d=$(dirname `readlink -f $0`)
-if [ $# -lt 1 ]; then
-	echo "Extracts unique words and their frequencies from a subset of a corpus."
-	echo
-	echo "Usage: `basename $0` input_file [number_of_lines] > output_file"
-	echo -e "\tinput_file contains a sentence per line."
-	echo
-	echo "Script also removes words from the vocabulary if they contain a digit or a special character. Output is printed to stdout in a format suitable for use with Morfessor."
-	echo
-	exit
-fi
-
-srcname=$1
-reallen=0
-
-if [[ $# -gt 1 ]]; then
-  reallen=$2
-fi
-
-pattern_file=$d/invalid_vocab.patterns
-
-if [[ ! -f $pattern_file ]]; then
-  echo "Pattern file missing"
-  exit 1 
-fi
-
-#this awk strips entries from the vocabulary if they contain invalid characters
-#invalid characters are digits and punctuation marks, and words beginning or ending with a dash
-#uniq -c extracts the unique words and counts the occurrences
-
-if [[ $reallen -eq 0 ]]; then
-	#when a zero is passed, use the whole file
-  zcat -f $srcname | sed 's/ /\n/g' | egrep -v -f $pattern_file | sort | uniq -c | sed 's/^  *//' 
-
-else
-	zcat -f $srcname | head -n $reallen | sed 's/ /\n/g' | egrep -v -f $pattern_file | sort | uniq -c | sed 's/^  *//'
-fi
-
author	Chris Dyer <cdyer@cs.cmu.edu>	2012-10-11 14:06:32 -0400
committer	Chris Dyer <cdyer@cs.cmu.edu>	2012-10-11 14:06:32 -0400
commit	07ea7b64b6f85e5798a8068453ed9fd2b97396db (patch)
tree	644496a1690d84d82a396bbc1e39160788beb2cd /gi/morf-segmentation/vocabextractor.sh
parent	37b9e45e5cb29d708f7249dbe0b0fb27685282a0 (diff)
parent	a36fcc5d55c1de84ae68c1091ebff2b1c32dc3b7 (diff)