From 4223261682388944fe1b1cf31b9d51d88f9ad53b Mon Sep 17 00:00:00 2001 From: Patrick Simianer
Date: Thu, 26 Feb 2015 13:26:37 +0100
Subject: refactoring
---
training/dtrain/examples/parallelized/README | 2 +-
training/dtrain/examples/parallelized/dtrain.ini | 13 +---
training/dtrain/examples/parallelized/work/out.0.0 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.0.1 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.0.2 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.1.0 | 72 +++++++-------------
training/dtrain/examples/parallelized/work/out.1.1 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.1.2 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.2.0 | 72 +++++++-------------
training/dtrain/examples/parallelized/work/out.2.1 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.2.2 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.3.0 | 74 +++++++-------------
training/dtrain/examples/parallelized/work/out.3.1 | 72 +++++++-------------
training/dtrain/examples/parallelized/work/out.3.2 | 72 +++++++-------------
.../dtrain/examples/parallelized/work/shard.0.0.in | 4 +-
.../dtrain/examples/parallelized/work/shard.1.0.in | 4 +-
.../dtrain/examples/parallelized/work/shard.2.0.in | 4 +-
.../dtrain/examples/parallelized/work/shard.3.0.in | 2 +-
.../dtrain/examples/parallelized/work/weights.0 | 24 +++----
.../dtrain/examples/parallelized/work/weights.0.0 | 23 +++----
.../dtrain/examples/parallelized/work/weights.0.1 | 24 +++----
.../dtrain/examples/parallelized/work/weights.0.2 | 24 +++----
.../dtrain/examples/parallelized/work/weights.1 | 24 +++----
.../dtrain/examples/parallelized/work/weights.1.0 | 23 ++++---
.../dtrain/examples/parallelized/work/weights.1.1 | 24 +++----
.../dtrain/examples/parallelized/work/weights.1.2 | 24 +++----
.../dtrain/examples/parallelized/work/weights.2 | 24 +++----
.../dtrain/examples/parallelized/work/weights.2.0 | 22 +++---
.../dtrain/examples/parallelized/work/weights.2.1 | 24 +++----
.../dtrain/examples/parallelized/work/weights.2.2 | 24 +++----
.../dtrain/examples/parallelized/work/weights.3.0 | 24 +++----
.../dtrain/examples/parallelized/work/weights.3.1 | 24 +++----
.../dtrain/examples/parallelized/work/weights.3.2 | 24 +++----
training/dtrain/examples/standard/dtrain.ini | 29 ++------
training/dtrain/examples/toy/dtrain.ini | 10 +--
training/dtrain/examples/toy/expected-output | 79 +++++++---------------
training/dtrain/examples/toy/weights | 4 ++
37 files changed, 534 insertions(+), 853 deletions(-)
create mode 100644 training/dtrain/examples/toy/weights
(limited to 'training/dtrain/examples')
diff --git a/training/dtrain/examples/parallelized/README b/training/dtrain/examples/parallelized/README
index 2fb3b54e..c4addd81 100644
--- a/training/dtrain/examples/parallelized/README
+++ b/training/dtrain/examples/parallelized/README
@@ -1,5 +1,5 @@
run for example
- ../../parallelize.rb -c dtrain.ini -s 4 -e 3 -z -d ../../dtrain -p 2 -i in
+ ../../parallelize.rb -c dtrain.ini -s 4 -e 3 -d ../../dtrain -p 2 -i in
final weights will be in the file work/weights.2
diff --git a/training/dtrain/examples/parallelized/dtrain.ini b/training/dtrain/examples/parallelized/dtrain.ini
index 0b0932d6..9fc205a3 100644
--- a/training/dtrain/examples/parallelized/dtrain.ini
+++ b/training/dtrain/examples/parallelized/dtrain.ini
@@ -1,14 +1,7 @@
k=100
N=4
learning_rate=0.0001
-gamma=0
-loss_margin=1.0
-epochs=1
-scorer=stupid_bleu
-sample_from=kbest
-filter=uniq
-pair_sampling=XYX
-hi_lo=0.1
-select_weights=last
-print_weights=Glue WordPenalty LanguageModel LanguageModel_OOV PhraseModel_0 PhraseModel_1 PhraseModel_2 PhraseModel_3 PhraseModel_4 PhraseModel_5 PhraseModel_6 PassThrough
+error_margin=1.0
+iterations=1
decoder_config=cdec.ini
+print_weights=Glue WordPenalty LanguageModel LanguageModel_OOV PhraseModel_0 PhraseModel_1 PhraseModel_2 PhraseModel_3 PhraseModel_4 PhraseModel_5 PhraseModel_6 PassThrough
diff --git a/training/dtrain/examples/parallelized/work/out.0.0 b/training/dtrain/examples/parallelized/work/out.0.0
index 9154c906..77749404 100644
--- a/training/dtrain/examples/parallelized/work/out.0.0
+++ b/training/dtrain/examples/parallelized/work/out.0.0
@@ -1,65 +1,43 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 4087834873
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.0.0.in'
output 'work/weights.0.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = +0.257
- WordPenalty = +0.026926
- LanguageModel = +0.67342
- LanguageModel_OOV = -0.046
- PhraseModel_0 = +0.25329
- PhraseModel_1 = +0.20036
- PhraseModel_2 = +0.00060731
- PhraseModel_3 = +0.65578
- PhraseModel_4 = +0.47916
- PhraseModel_5 = +0.004
- PhraseModel_6 = +0.1829
- PassThrough = -0.082
+ Glue = +0.3404
+ WordPenalty = -0.017632
+ LanguageModel = +0.72958
+ LanguageModel_OOV = -0.235
+ PhraseModel_0 = -0.43721
+ PhraseModel_1 = +1.01
+ PhraseModel_2 = +1.3525
+ PhraseModel_3 = -0.25541
+ PhraseModel_4 = -0.78115
+ PhraseModel_5 = +0
+ PhraseModel_6 = -0.3681
+ PassThrough = -0.3304
---
- 1best avg score: 0.04518 (+0.04518)
- 1best avg model score: 32.803 (+32.803)
- avg # pairs: 1266.3
- avg # rank err: 857
- avg # margin viol: 386.67
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.19474 (+0.19474)
+ 1best avg model score: 0.52232
+ avg # pairs: 2513
+ non-0 feature count: 11
avg list sz: 100
- avg f count: 10.853
-(time 0.47 min, 9.3 s/S)
-
-Writing weights file to 'work/weights.0.0' ...
-done
+ avg f count: 11.42
+(time 0.32 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.04518].
-This took 0.46667 min.
+Best iteration: 1 [GOLD = 0.19474].
+This took 0.31667 min.
diff --git a/training/dtrain/examples/parallelized/work/out.0.1 b/training/dtrain/examples/parallelized/work/out.0.1
index 0dbc7bd3..d0dee623 100644
--- a/training/dtrain/examples/parallelized/work/out.0.1
+++ b/training/dtrain/examples/parallelized/work/out.0.1
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 2283043509
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.0.0.in'
output 'work/weights.0.1'
weights in 'work/weights.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.17905
- WordPenalty = +0.062126
- LanguageModel = +0.66825
- LanguageModel_OOV = -0.15248
- PhraseModel_0 = -0.55811
- PhraseModel_1 = +0.12741
- PhraseModel_2 = +0.60388
- PhraseModel_3 = -0.44464
- PhraseModel_4 = -0.63137
- PhraseModel_5 = -0.0084
- PhraseModel_6 = -0.20165
- PassThrough = -0.23468
+ Glue = -0.40908
+ WordPenalty = +0.12967
+ LanguageModel = +0.39892
+ LanguageModel_OOV = -0.6314
+ PhraseModel_0 = -0.63992
+ PhraseModel_1 = +0.74198
+ PhraseModel_2 = +1.3096
+ PhraseModel_3 = -0.1216
+ PhraseModel_4 = -1.2274
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.21093
+ PassThrough = -0.66155
---
- 1best avg score: 0.14066 (+0.14066)
- 1best avg model score: -37.614 (-37.614)
- avg # pairs: 1244.7
- avg # rank err: 728
- avg # margin viol: 516.67
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.15735 (+0.15735)
+ 1best avg model score: 46.831
+ avg # pairs: 2132.3
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 11.507
-(time 0.45 min, 9 s/S)
-
-Writing weights file to 'work/weights.0.1' ...
-done
+ avg f count: 10.64
+(time 0.38 min, 7 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.14066].
-This took 0.45 min.
+Best iteration: 1 [GOLD = 0.15735].
+This took 0.38333 min.
diff --git a/training/dtrain/examples/parallelized/work/out.0.2 b/training/dtrain/examples/parallelized/work/out.0.2
index fcecc7e1..9c4b110b 100644
--- a/training/dtrain/examples/parallelized/work/out.0.2
+++ b/training/dtrain/examples/parallelized/work/out.0.2
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 3693132895
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.0.0.in'
output 'work/weights.0.2'
weights in 'work/weights.1'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.019275
- WordPenalty = +0.022192
- LanguageModel = +0.40688
- LanguageModel_OOV = -0.36397
- PhraseModel_0 = -0.36273
- PhraseModel_1 = +0.56432
- PhraseModel_2 = +0.85638
- PhraseModel_3 = -0.20222
- PhraseModel_4 = -0.48295
- PhraseModel_5 = +0.03145
- PhraseModel_6 = -0.26092
- PassThrough = -0.38122
+ Glue = -0.44422
+ WordPenalty = +0.1032
+ LanguageModel = +0.66474
+ LanguageModel_OOV = -0.62252
+ PhraseModel_0 = -0.59993
+ PhraseModel_1 = +0.78992
+ PhraseModel_2 = +1.3149
+ PhraseModel_3 = +0.21434
+ PhraseModel_4 = -1.0174
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.18452
+ PassThrough = -0.65268
---
- 1best avg score: 0.18982 (+0.18982)
- 1best avg model score: 1.7096 (+1.7096)
- avg # pairs: 1524.3
- avg # rank err: 813.33
- avg # margin viol: 702.67
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.24722 (+0.24722)
+ 1best avg model score: 61.971
+ avg # pairs: 2017.7
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 11.32
-(time 0.53 min, 11 s/S)
-
-Writing weights file to 'work/weights.0.2' ...
-done
+ avg f count: 10.42
+(time 0.3 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.18982].
-This took 0.53333 min.
+Best iteration: 1 [GOLD = 0.24722].
+This took 0.3 min.
diff --git a/training/dtrain/examples/parallelized/work/out.1.0 b/training/dtrain/examples/parallelized/work/out.1.0
index 595dfc94..3dc4dca6 100644
--- a/training/dtrain/examples/parallelized/work/out.1.0
+++ b/training/dtrain/examples/parallelized/work/out.1.0
@@ -1,65 +1,43 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 859043351
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.1.0.in'
output 'work/weights.1.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.3229
- WordPenalty = +0.27969
- LanguageModel = +1.3645
- LanguageModel_OOV = -0.0443
- PhraseModel_0 = -0.19049
- PhraseModel_1 = -0.077698
- PhraseModel_2 = +0.058898
- PhraseModel_3 = +0.017251
- PhraseModel_4 = -1.5474
- PhraseModel_5 = +0
- PhraseModel_6 = -0.1818
- PassThrough = -0.193
+ Glue = -0.2722
+ WordPenalty = +0.05433
+ LanguageModel = +0.69948
+ LanguageModel_OOV = -0.2641
+ PhraseModel_0 = -1.4208
+ PhraseModel_1 = -1.563
+ PhraseModel_2 = -0.21051
+ PhraseModel_3 = -0.17764
+ PhraseModel_4 = -1.6583
+ PhraseModel_5 = +0.0794
+ PhraseModel_6 = +0.1528
+ PassThrough = -0.2367
---
- 1best avg score: 0.070229 (+0.070229)
- 1best avg model score: -44.01 (-44.01)
- avg # pairs: 1294
- avg # rank err: 878.67
- avg # margin viol: 350.67
- k-best loss imp: 100%
- non0 feature count: 11
+ 1best avg score: 0.071329 (+0.071329)
+ 1best avg model score: -41.362
+ avg # pairs: 1862.3
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 11.487
-(time 0.28 min, 5.7 s/S)
-
-Writing weights file to 'work/weights.1.0' ...
-done
+ avg f count: 11.847
+(time 0.28 min, 5 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.070229].
+Best iteration: 1 [GOLD = 0.071329].
This took 0.28333 min.
diff --git a/training/dtrain/examples/parallelized/work/out.1.1 b/training/dtrain/examples/parallelized/work/out.1.1
index 9346fc82..79ac35dc 100644
--- a/training/dtrain/examples/parallelized/work/out.1.1
+++ b/training/dtrain/examples/parallelized/work/out.1.1
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 3557309480
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.1.0.in'
output 'work/weights.1.1'
weights in 'work/weights.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.26425
- WordPenalty = +0.047881
- LanguageModel = +0.78496
- LanguageModel_OOV = -0.49307
- PhraseModel_0 = -0.58703
- PhraseModel_1 = -0.33425
- PhraseModel_2 = +0.20834
- PhraseModel_3 = -0.043346
- PhraseModel_4 = -0.60761
- PhraseModel_5 = +0.123
- PhraseModel_6 = -0.05415
- PassThrough = -0.42167
+ Glue = -0.20488
+ WordPenalty = -0.0091745
+ LanguageModel = +0.79433
+ LanguageModel_OOV = -0.4309
+ PhraseModel_0 = -0.56242
+ PhraseModel_1 = +0.85363
+ PhraseModel_2 = +1.3458
+ PhraseModel_3 = -0.13095
+ PhraseModel_4 = -0.94762
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.16003
+ PassThrough = -0.46105
---
- 1best avg score: 0.085952 (+0.085952)
- 1best avg model score: -45.175 (-45.175)
- avg # pairs: 1180.7
- avg # rank err: 668.33
- avg # margin viol: 512.33
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.13017 (+0.13017)
+ 1best avg model score: 14.53
+ avg # pairs: 1968
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 12
-(time 0.27 min, 5.3 s/S)
-
-Writing weights file to 'work/weights.1.1' ...
-done
+ avg f count: 11
+(time 0.33 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.085952].
-This took 0.26667 min.
+Best iteration: 1 [GOLD = 0.13017].
+This took 0.33333 min.
diff --git a/training/dtrain/examples/parallelized/work/out.1.2 b/training/dtrain/examples/parallelized/work/out.1.2
index 08f07a75..8c4f8b03 100644
--- a/training/dtrain/examples/parallelized/work/out.1.2
+++ b/training/dtrain/examples/parallelized/work/out.1.2
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 56743915
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.1.0.in'
output 'work/weights.1.2'
weights in 'work/weights.1'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.23608
- WordPenalty = +0.10931
- LanguageModel = +0.81339
- LanguageModel_OOV = -0.33238
- PhraseModel_0 = -0.53685
- PhraseModel_1 = -0.049658
- PhraseModel_2 = +0.40277
- PhraseModel_3 = +0.14601
- PhraseModel_4 = -0.72851
- PhraseModel_5 = +0.03475
- PhraseModel_6 = -0.27192
- PassThrough = -0.34763
+ Glue = -0.49853
+ WordPenalty = +0.07636
+ LanguageModel = +1.3183
+ LanguageModel_OOV = -0.60902
+ PhraseModel_0 = -0.22481
+ PhraseModel_1 = +0.86369
+ PhraseModel_2 = +1.0747
+ PhraseModel_3 = +0.18002
+ PhraseModel_4 = -0.84661
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = +0.11247
+ PassThrough = -0.63918
---
- 1best avg score: 0.10073 (+0.10073)
- 1best avg model score: -38.422 (-38.422)
- avg # pairs: 1505.3
- avg # rank err: 777
- avg # margin viol: 691.67
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.15478 (+0.15478)
+ 1best avg model score: -7.2154
+ avg # pairs: 1776
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 12
-(time 0.35 min, 7 s/S)
-
-Writing weights file to 'work/weights.1.2' ...
-done
+ avg f count: 11.327
+(time 0.27 min, 5 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.10073].
-This took 0.35 min.
+Best iteration: 1 [GOLD = 0.15478].
+This took 0.26667 min.
diff --git a/training/dtrain/examples/parallelized/work/out.2.0 b/training/dtrain/examples/parallelized/work/out.2.0
index 25ef6d4e..07c85963 100644
--- a/training/dtrain/examples/parallelized/work/out.2.0
+++ b/training/dtrain/examples/parallelized/work/out.2.0
@@ -1,65 +1,43 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 2662215673
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.2.0.in'
output 'work/weights.2.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.1259
- WordPenalty = +0.048294
- LanguageModel = +0.36254
- LanguageModel_OOV = -0.1228
- PhraseModel_0 = +0.26357
- PhraseModel_1 = +0.24793
- PhraseModel_2 = +0.0063763
- PhraseModel_3 = -0.18966
- PhraseModel_4 = -0.226
+ Glue = -0.2109
+ WordPenalty = +0.14922
+ LanguageModel = +0.79686
+ LanguageModel_OOV = -0.6627
+ PhraseModel_0 = +0.37999
+ PhraseModel_1 = +0.69213
+ PhraseModel_2 = +0.3422
+ PhraseModel_3 = +1.1426
+ PhraseModel_4 = -0.55413
PhraseModel_5 = +0
- PhraseModel_6 = +0.0743
- PassThrough = -0.1335
+ PhraseModel_6 = +0.0676
+ PassThrough = -0.6343
---
- 1best avg score: 0.072836 (+0.072836)
- 1best avg model score: -0.56296 (-0.56296)
- avg # pairs: 1094.7
- avg # rank err: 658
- avg # margin viol: 436.67
- k-best loss imp: 100%
- non0 feature count: 11
+ 1best avg score: 0.072374 (+0.072374)
+ 1best avg model score: -27.384
+ avg # pairs: 2582
+ non-0 feature count: 11
avg list sz: 100
- avg f count: 10.813
-(time 0.13 min, 2.7 s/S)
-
-Writing weights file to 'work/weights.2.0' ...
-done
+ avg f count: 11.54
+(time 0.32 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.072836].
-This took 0.13333 min.
+Best iteration: 1 [GOLD = 0.072374].
+This took 0.31667 min.
diff --git a/training/dtrain/examples/parallelized/work/out.2.1 b/training/dtrain/examples/parallelized/work/out.2.1
index 8e4efde9..c54bb1b1 100644
--- a/training/dtrain/examples/parallelized/work/out.2.1
+++ b/training/dtrain/examples/parallelized/work/out.2.1
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 3092904479
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.2.0.in'
output 'work/weights.2.1'
weights in 'work/weights.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.10385
- WordPenalty = +0.038717
- LanguageModel = +0.49413
- LanguageModel_OOV = -0.24887
- PhraseModel_0 = -0.32102
- PhraseModel_1 = +0.34413
- PhraseModel_2 = +0.62366
- PhraseModel_3 = -0.49337
- PhraseModel_4 = -0.77005
- PhraseModel_5 = +0.007
- PhraseModel_6 = -0.05055
- PassThrough = -0.23928
+ Glue = -0.76608
+ WordPenalty = +0.15938
+ LanguageModel = +1.5897
+ LanguageModel_OOV = -0.521
+ PhraseModel_0 = -0.58348
+ PhraseModel_1 = +0.29828
+ PhraseModel_2 = +0.78493
+ PhraseModel_3 = +0.083222
+ PhraseModel_4 = -0.93843
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.27382
+ PassThrough = -0.55115
---
- 1best avg score: 0.10245 (+0.10245)
- 1best avg model score: -20.384 (-20.384)
- avg # pairs: 1741.7
- avg # rank err: 953.67
- avg # margin viol: 585.33
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.12881 (+0.12881)
+ 1best avg model score: -9.6731
+ avg # pairs: 2020.3
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 11.977
-(time 0.12 min, 2.3 s/S)
-
-Writing weights file to 'work/weights.2.1' ...
-done
+ avg f count: 12
+(time 0.32 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.10245].
-This took 0.11667 min.
+Best iteration: 1 [GOLD = 0.12881].
+This took 0.31667 min.
diff --git a/training/dtrain/examples/parallelized/work/out.2.2 b/training/dtrain/examples/parallelized/work/out.2.2
index e0ca2110..f5d6229f 100644
--- a/training/dtrain/examples/parallelized/work/out.2.2
+++ b/training/dtrain/examples/parallelized/work/out.2.2
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 2803362953
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.2.0.in'
output 'work/weights.2.2'
weights in 'work/weights.1'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 3
+ .... 3
WEIGHTS
- Glue = -0.32907
- WordPenalty = +0.049596
- LanguageModel = +0.33496
- LanguageModel_OOV = -0.44357
- PhraseModel_0 = -0.3068
- PhraseModel_1 = +0.59376
- PhraseModel_2 = +0.86416
- PhraseModel_3 = -0.21072
- PhraseModel_4 = -0.65734
- PhraseModel_5 = +0.03475
- PhraseModel_6 = -0.10653
- PassThrough = -0.46082
+ Glue = -0.90863
+ WordPenalty = +0.10819
+ LanguageModel = +0.5239
+ LanguageModel_OOV = -0.41623
+ PhraseModel_0 = -0.86868
+ PhraseModel_1 = +0.40784
+ PhraseModel_2 = +1.1793
+ PhraseModel_3 = -0.24698
+ PhraseModel_4 = -1.2353
+ PhraseModel_5 = +0.03375
+ PhraseModel_6 = -0.17883
+ PassThrough = -0.44638
---
- 1best avg score: 0.25055 (+0.25055)
- 1best avg model score: -1.4459 (-1.4459)
- avg # pairs: 1689
- avg # rank err: 755.67
- avg # margin viol: 829.33
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.12788 (+0.12788)
+ 1best avg model score: 41.302
+ avg # pairs: 2246.3
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 10.53
-(time 0.13 min, 2.7 s/S)
-
-Writing weights file to 'work/weights.2.2' ...
-done
+ avg f count: 10.98
+(time 0.35 min, 7 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.25055].
-This took 0.13333 min.
+Best iteration: 1 [GOLD = 0.12788].
+This took 0.35 min.
diff --git a/training/dtrain/examples/parallelized/work/out.3.0 b/training/dtrain/examples/parallelized/work/out.3.0
index 3c074f04..fa499523 100644
--- a/training/dtrain/examples/parallelized/work/out.3.0
+++ b/training/dtrain/examples/parallelized/work/out.3.0
@@ -1,65 +1,43 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 316107185
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.3.0.in'
output 'work/weights.3.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 1
+ .. 1
WEIGHTS
- Glue = +0.046
- WordPenalty = +0.17328
- LanguageModel = +1.1667
- LanguageModel_OOV = +0.066
- PhraseModel_0 = -1.1694
- PhraseModel_1 = -0.9883
- PhraseModel_2 = +0.036205
- PhraseModel_3 = -0.77387
- PhraseModel_4 = -1.5019
- PhraseModel_5 = +0.024
- PhraseModel_6 = -0.514
- PassThrough = +0.031
+ Glue = -0.09
+ WordPenalty = +0.32442
+ LanguageModel = +2.5769
+ LanguageModel_OOV = -0.009
+ PhraseModel_0 = -0.58972
+ PhraseModel_1 = +0.063691
+ PhraseModel_2 = +0.5366
+ PhraseModel_3 = +0.12867
+ PhraseModel_4 = -1.9801
+ PhraseModel_5 = +0.018
+ PhraseModel_6 = -0.486
+ PassThrough = -0.09
---
- 1best avg score: 0.032916 (+0.032916)
- 1best avg model score: 0 (+0)
- avg # pairs: 900
- avg # rank err: 900
- avg # margin viol: 0
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.034204 (+0.034204)
+ 1best avg model score: 0
+ avg # pairs: 1700
+ non-0 feature count: 12
avg list sz: 100
- avg f count: 11.72
-(time 0.23 min, 14 s/S)
-
-Writing weights file to 'work/weights.3.0' ...
-done
+ avg f count: 10.8
+(time 0.1 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.032916].
-This took 0.23333 min.
+Best iteration: 1 [GOLD = 0.034204].
+This took 0.1 min.
diff --git a/training/dtrain/examples/parallelized/work/out.3.1 b/training/dtrain/examples/parallelized/work/out.3.1
index 241d3455..c4b3aa3c 100644
--- a/training/dtrain/examples/parallelized/work/out.3.1
+++ b/training/dtrain/examples/parallelized/work/out.3.1
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 353677750
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.3.0.in'
output 'work/weights.3.1'
weights in 'work/weights.0'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 1
+ .. 1
WEIGHTS
- Glue = -0.08475
- WordPenalty = +0.11151
- LanguageModel = +1.0635
- LanguageModel_OOV = -0.11468
- PhraseModel_0 = -0.062922
- PhraseModel_1 = +0.0035552
- PhraseModel_2 = +0.039692
- PhraseModel_3 = +0.080265
- PhraseModel_4 = -0.57787
- PhraseModel_5 = +0.0174
- PhraseModel_6 = -0.17095
- PassThrough = -0.18248
+ Glue = +0.31832
+ WordPenalty = +0.11139
+ LanguageModel = +0.95438
+ LanguageModel_OOV = -0.0608
+ PhraseModel_0 = -0.98113
+ PhraseModel_1 = -0.090531
+ PhraseModel_2 = +0.79088
+ PhraseModel_3 = -0.57623
+ PhraseModel_4 = -1.4382
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.10812
+ PassThrough = -0.09095
---
- 1best avg score: 0.16117 (+0.16117)
- 1best avg model score: -67.89 (-67.89)
- avg # pairs: 1411
- avg # rank err: 460
- avg # margin viol: 951
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.084989 (+0.084989)
+ 1best avg model score: -52.323
+ avg # pairs: 2487
+ non-0 feature count: 12
avg list sz: 100
avg f count: 12
-(time 0.22 min, 13 s/S)
-
-Writing weights file to 'work/weights.3.1' ...
-done
+(time 0.1 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.16117].
-This took 0.21667 min.
+Best iteration: 1 [GOLD = 0.084989].
+This took 0.1 min.
diff --git a/training/dtrain/examples/parallelized/work/out.3.2 b/training/dtrain/examples/parallelized/work/out.3.2
index b995daf5..eb27dac2 100644
--- a/training/dtrain/examples/parallelized/work/out.3.2
+++ b/training/dtrain/examples/parallelized/work/out.3.2
@@ -1,66 +1,44 @@
- cdec cfg 'cdec.ini'
Loading the LM will be faster if you build a binary file.
Reading ../standard/nc-wmt11.en.srilm.gz
----5---10---15---20---25---30---35---40---45---50---55---60---65---70---75---80---85---90---95--100
****************************************************************************************************
-Seeding random number sequence to 3001145976
-
dtrain
Parameters:
k 100
N 4
T 1
- batch 0
- scorer 'stupid_bleu'
- sample from 'kbest'
- filter 'uniq'
learning rate 0.0001
- gamma 0
- loss margin 1
- faster perceptron 0
- pairs 'XYX'
- hi lo 0.1
- pair threshold 0
- select weights 'last'
- l1 reg 0 'none'
- pclr no
- max pairs 4294967295
- repeat 1
- cdec cfg 'cdec.ini'
- input ''
+ error margin 1
+ l1 reg 0
+ decoder conf 'cdec.ini'
+ input 'work/shard.3.0.in'
output 'work/weights.3.2'
weights in 'work/weights.1'
-(a dot represents 10 inputs)
+(a dot per input)
Iteration #1 of 1.
- 1
+ .. 1
WEIGHTS
- Glue = -0.13247
- WordPenalty = +0.053592
- LanguageModel = +0.72105
- LanguageModel_OOV = -0.30827
- PhraseModel_0 = -0.37053
- PhraseModel_1 = +0.17551
- PhraseModel_2 = +0.5
- PhraseModel_3 = -0.1459
- PhraseModel_4 = -0.59563
- PhraseModel_5 = +0.03475
- PhraseModel_6 = -0.11143
- PassThrough = -0.32553
+ Glue = -0.12993
+ WordPenalty = +0.13651
+ LanguageModel = +0.58946
+ LanguageModel_OOV = -0.48362
+ PhraseModel_0 = -0.81262
+ PhraseModel_1 = +0.44273
+ PhraseModel_2 = +1.1733
+ PhraseModel_3 = -0.1826
+ PhraseModel_4 = -1.2213
+ PhraseModel_5 = +0.02435
+ PhraseModel_6 = -0.18823
+ PassThrough = -0.51378
---
- 1best avg score: 0.12501 (+0.12501)
- 1best avg model score: -62.128 (-62.128)
- avg # pairs: 979
- avg # rank err: 539
- avg # margin viol: 440
- k-best loss imp: 100%
- non0 feature count: 12
+ 1best avg score: 0.12674 (+0.12674)
+ 1best avg model score: -7.2878
+ avg # pairs: 1769
+ non-0 feature count: 12
avg list sz: 100
avg f count: 12
-(time 0.22 min, 13 s/S)
-
-Writing weights file to 'work/weights.3.2' ...
-done
+(time 0.1 min, 6 s/S)
---
-Best iteration: 1 [SCORE 'stupid_bleu'=0.12501].
-This took 0.21667 min.
+Best iteration: 1 [GOLD = 0.12674].
+This took 0.1 min.
diff --git a/training/dtrain/examples/parallelized/work/shard.0.0.in b/training/dtrain/examples/parallelized/work/shard.0.0.in
index d1b48321..a0ef6f54 100644
--- a/training/dtrain/examples/parallelized/work/shard.0.0.in
+++ b/training/dtrain/examples/parallelized/work/shard.0.0.in
@@ -1,3 +1,3 @@
-