diff options
author | Paul Baltescu <pauldb89@gmail.com> | 2013-02-21 14:13:55 +0000 |
---|---|---|
committer | Paul Baltescu <pauldb89@gmail.com> | 2013-02-21 14:13:55 +0000 |
commit | bca26d953a774b8efca12f30407390b3f5eef9d0 (patch) | |
tree | fe922de5c89b1844f677d550dcc24e87edd67a55 /python/tests/extractor/run.sh | |
parent | 54a1c0e2bde259e3acc9c0a8ec8da3c7704e80ca (diff) | |
parent | 95c364f2cb002241c4a62bedb1c5ef6f1e9a7f22 (diff) |
Merge branch 'master' of https://github.com/pauldb89/cdec
Diffstat (limited to 'python/tests/extractor/run.sh')
-rwxr-xr-x | python/tests/extractor/run.sh | 14 |
1 files changed, 14 insertions, 0 deletions
diff --git a/python/tests/extractor/run.sh b/python/tests/extractor/run.sh new file mode 100755 index 00000000..f44da9f8 --- /dev/null +++ b/python/tests/extractor/run.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash + +# Make sure that the sa and online extractors are producing the same (correct) output + +set -x verbose + +python -m cdec.sa.compile -a corpus.al.gz -b corpus.fr-en.gz -o extract >| extract.ini + +cat test.in | python -m cdec.sa.extract -c extract.ini -g gold -o 2>&1 | egrep '\[X\].+\|\|\|.+\|\|\|.+\|\|\|.+\|\|\|'|sed -re 's/INFO.+://g' | ./refmt.py | LC_ALL=C sort >| rules.sort + +cd gold && cat grammar.0|sed -re 's/Egiv.+(IsSingletonF=)/\1/g'|LC_ALL=C sort >| rules.sort && cd .. + +diff gold/rules.sort gold-rules.sort +diff rules.sort gold-rules.sort |