summaryrefslogtreecommitdiff
path: root/corpus
diff options
context:
space:
mode:
authorChris Dyer <cdyer@cs.cmu.edu>2012-10-25 16:05:56 -0400
committerChris Dyer <cdyer@cs.cmu.edu>2012-10-25 16:05:56 -0400
commit198cc8c885a69b9bbe06c842d3a868583885c526 (patch)
treeddc3093208c38c1288a581968226dc3c8133a7fc /corpus
parent500cf9773a88993ea4fa3d17e7f886dfcf7a546e (diff)
add self translation
Diffstat (limited to 'corpus')
-rwxr-xr-xcorpus/add-self-translations.pl29
1 files changed, 29 insertions, 0 deletions
diff --git a/corpus/add-self-translations.pl b/corpus/add-self-translations.pl
new file mode 100755
index 00000000..153bc454
--- /dev/null
+++ b/corpus/add-self-translations.pl
@@ -0,0 +1,29 @@
+#!/usr/bin/perl -w
+use strict;
+
+# ADDS SELF-TRANSLATIONS OF POORLY ATTESTED WORDS TO THE PARALLEL DATA
+
+my %df;
+my %def;
+while(<>) {
+ print;
+ chomp;
+ my ($sf, $se) = split / \|\|\| /;
+ die "Format error: $_\n" unless defined $sf && defined $se;
+ my @fs = split /\s+/, $sf;
+ my @es = split /\s+/, $se;
+ for my $f (@fs) {
+ $df{$f}++;
+ for my $e (@es) {
+ if ($f eq $e) { $def{$f}++; }
+ }
+ }
+}
+
+for my $k (sort keys %def) {
+ next if $df{$k} > 4;
+ print "$k ||| $k\n";
+ print "$k ||| $k\n";
+ print "$k ||| $k\n";
+}
+