From 6ac91aa63cd0bce3a9a4fcc781d3ad66a326d5c8 Mon Sep 17 00:00:00 2001
From: Chris Dyer <redpony@gmail.com>
Date: Thu, 9 Oct 2014 14:14:33 -0400
Subject: fix header names in util/

---
 utils/alias_sampler.h        | 4 ++--
 utils/alignment_io.h         | 4 ++--
 utils/b64tools.h             | 4 ++--
 utils/corpus_tools.h         | 4 ++--
 utils/exp_semiring.h         | 4 ++--
 utils/fast_sparse_vector.h   | 4 ++--
 utils/fdict.h                | 4 ++--
 utils/feature_vector.h       | 4 ++--
 utils/filelib.h              | 4 ++--
 utils/kernel_string_subseq.h | 4 ++--
 utils/m.h                    | 4 ++--
 utils/murmur_hash3.h         | 4 ++--
 utils/perfect_hash.h         | 4 ++--
 utils/prob.h                 | 4 ++--
 utils/small_vector.h         | 4 ++--
 utils/sparse_vector.h        | 4 ++--
 utils/star.h                 | 4 ++--
 utils/tdict.h                | 4 ++--
 utils/timing_stats.h         | 4 ++--
 utils/verbose.h              | 4 ++--
 utils/weights.h              | 4 ++--
 utils/wordid.h               | 4 ++--
 22 files changed, 44 insertions(+), 44 deletions(-)

(limited to 'utils')
diff --git a/utils/alias_sampler.h b/utils/alias_sampler.h
index 81541f7a..0f9d3f6d 100644
--- a/utils/alias_sampler.h
+++ b/utils/alias_sampler.h
@@ -1,5 +1,5 @@
-#ifndef _ALIAS_SAMPLER_H_
-#define _ALIAS_SAMPLER_H_
+#ifndef ALIAS_SAMPLER_H_
+#define ALIAS_SAMPLER_H_
 
 #include <vector>
 #include <limits>
diff --git a/utils/alignment_io.h b/utils/alignment_io.h
index 63fb916b..ec70688e 100644
--- a/utils/alignment_io.h
+++ b/utils/alignment_io.h
@@ -1,5 +1,5 @@
-#ifndef _ALIGNMENT_IO_H_
-#define _ALIGNMENT_IO_H_
+#ifndef ALIGNMENT_IO_H_
+#define ALIGNMENT_IO_H_
 
 #include <string>
 #include <iostream>
diff --git a/utils/b64tools.h b/utils/b64tools.h
index c821fc8f..130a9102 100644
--- a/utils/b64tools.h
+++ b/utils/b64tools.h
@@ -1,5 +1,5 @@
-#ifndef _B64_TOOLS_H_
-#define _B64_TOOLS_H_
+#ifndef B64_TOOLS_H_
+#define B64_TOOLS_H_
 
 namespace B64 {
   bool b64decode(const unsigned char* data, const size_t insize, char* out, const size_t outsize);
diff --git a/utils/corpus_tools.h b/utils/corpus_tools.h
index f6699d87..3ccaf6ef 100644
--- a/utils/corpus_tools.h
+++ b/utils/corpus_tools.h
@@ -1,5 +1,5 @@
-#ifndef _CORPUS_TOOLS_H_
-#define _CORPUS_TOOLS_H_
+#ifndef CORPUS_TOOLS_H_
+#define CORPUS_TOOLS_H_
 
 #include <string>
 #include <set>
diff --git a/utils/exp_semiring.h b/utils/exp_semiring.h
index 26a22071..164286e3 100644
--- a/utils/exp_semiring.h
+++ b/utils/exp_semiring.h
@@ -1,5 +1,5 @@
-#ifndef _EXP_SEMIRING_H_
-#define _EXP_SEMIRING_H_
+#ifndef EXP_SEMIRING_H_
+#define EXP_SEMIRING_H_
 
 #include <iostream>
 #include "star.h"
diff --git a/utils/fast_sparse_vector.h b/utils/fast_sparse_vector.h
index 6e2a77cd..1e0ab428 100644
--- a/utils/fast_sparse_vector.h
+++ b/utils/fast_sparse_vector.h
@@ -1,5 +1,5 @@
-#ifndef _FAST_SPARSE_VECTOR_H_
-#define _FAST_SPARSE_VECTOR_H_
+#ifndef FAST_SPARSE_VECTOR_H_
+#define FAST_SPARSE_VECTOR_H_
 
 // FastSparseVector<T> is a integer indexed unordered map that supports very fast
 // (mathematical) vector operations when the sizes are very small, and reasonably
diff --git a/utils/fdict.h b/utils/fdict.h
index eb853fb2..94763890 100644
--- a/utils/fdict.h
+++ b/utils/fdict.h
@@ -1,5 +1,5 @@
-#ifndef _FDICT_H_
-#define _FDICT_H_
+#ifndef FDICT_H_
+#define FDICT_H_
 
 #ifdef HAVE_CONFIG_H
 #include "config.h"
diff --git a/utils/feature_vector.h b/utils/feature_vector.h
index a7b61a66..bf77b5ac 100644
--- a/utils/feature_vector.h
+++ b/utils/feature_vector.h
@@ -1,5 +1,5 @@
-#ifndef _FEATURE_VECTOR_H_
-#define _FEATURE_VECTOR_H_
+#ifndef FEATURE_VECTOR_H_
+#define FEATURE_VECTOR_H_
 
 #include <vector>
 #include "sparse_vector.h"
diff --git a/utils/filelib.h b/utils/filelib.h
index 4fa69760..90620d05 100644
--- a/utils/filelib.h
+++ b/utils/filelib.h
@@ -1,5 +1,5 @@
-#ifndef _FILELIB_H_
-#define _FILELIB_H_
+#ifndef FILELIB_H_
+#define FILELIB_H_
 
 #include <cassert>
 #include <string>
diff --git a/utils/kernel_string_subseq.h b/utils/kernel_string_subseq.h
index 516e8b89..00ee7da7 100644
--- a/utils/kernel_string_subseq.h
+++ b/utils/kernel_string_subseq.h
@@ -1,5 +1,5 @@
-#ifndef _KERNEL_STRING_SUBSEQ_H_
-#define _KERNEL_STRING_SUBSEQ_H_
+#ifndef KERNEL_STRING_SUBSEQ_H_
+#define KERNEL_STRING_SUBSEQ_H_
 
 #include <vector>
 #include <cmath>
diff --git a/utils/m.h b/utils/m.h
index dc881b36..bd82c305 100644
--- a/utils/m.h
+++ b/utils/m.h
@@ -1,5 +1,5 @@
-#ifndef _M_H_
-#define _M_H_
+#ifndef M_H_HEADER_
+#define M_H_HEADER_
 
 #include <cassert>
 #include <cmath>
diff --git a/utils/murmur_hash3.h b/utils/murmur_hash3.h
index a125d775..e8a8b10b 100644
--- a/utils/murmur_hash3.h
+++ b/utils/murmur_hash3.h
@@ -2,8 +2,8 @@
 // MurmurHash3 was written by Austin Appleby, and is placed in the public
 // domain. The author hereby disclaims copyright to this source code.
 
-#ifndef _MURMURHASH3_H_
-#define _MURMURHASH3_H_
+#ifndef MURMURHASH3_H_
+#define MURMURHASH3_H_
 
 //-----------------------------------------------------------------------------
 // Platform-specific functions and macros
diff --git a/utils/perfect_hash.h b/utils/perfect_hash.h
index 29ea48a9..8c12c9f0 100644
--- a/utils/perfect_hash.h
+++ b/utils/perfect_hash.h
@@ -1,5 +1,5 @@
-#ifndef _PERFECT_HASH_MAP_H_
-#define _PERFECT_HASH_MAP_H_
+#ifndef PERFECT_HASH_MAP_H_
+#define PERFECT_HASH_MAP_H_
 
 #include <vector>
 #include <boost/utility.hpp>
diff --git a/utils/prob.h b/utils/prob.h
index bc297870..32ba9a86 100644
--- a/utils/prob.h
+++ b/utils/prob.h
@@ -1,5 +1,5 @@
-#ifndef _PROB_H_
-#define _PROB_H_
+#ifndef PROB_H_
+#define PROB_H_
 
 #include "logval.h"
 
diff --git a/utils/small_vector.h b/utils/small_vector.h
index 280ab72c..c8cbcb2c 100644
--- a/utils/small_vector.h
+++ b/utils/small_vector.h
@@ -1,5 +1,5 @@
-#ifndef _SMALL_VECTOR_H_
-#define _SMALL_VECTOR_H_
+#ifndef SMALL_VECTOR_H_
+#define SMALL_VECTOR_H_
 
 /* REQUIRES that T is POD (can be memcpy).  won't work (yet) due to union with SMALL_VECTOR_POD==0 - may be possible to handle movable types that have ctor/dtor, by using  explicit allocation, ctor/dtor calls.  but for now JUST USE THIS FOR no-meaningful ctor/dtor POD types.
 
diff --git a/utils/sparse_vector.h b/utils/sparse_vector.h
index 049151f7..13601376 100644
--- a/utils/sparse_vector.h
+++ b/utils/sparse_vector.h
@@ -1,5 +1,5 @@
-#ifndef _SPARSE_VECTOR_H_
-#define _SPARSE_VECTOR_H_
+#ifndef SPARSE_VECTOR_H_
+#define SPARSE_VECTOR_H_
 
 #include "fast_sparse_vector.h"
 #define SparseVector FastSparseVector
diff --git a/utils/star.h b/utils/star.h
index 21977dc9..01433d12 100644
--- a/utils/star.h
+++ b/utils/star.h
@@ -1,5 +1,5 @@
-#ifndef _STAR_H_
-#define _STAR_H_
+#ifndef STAR_H_
+#define STAR_H_
 
 // star(x) computes the infinite sum x^0 + x^1 + x^2 + ...
 
diff --git a/utils/tdict.h b/utils/tdict.h
index bb19ecd5..eed33c3a 100644
--- a/utils/tdict.h
+++ b/utils/tdict.h
@@ -1,5 +1,5 @@
-#ifndef _TDICT_H_
-#define _TDICT_H_
+#ifndef TDICT_H_
+#define TDICT_H_
 
 #include <string>
 #include <vector>
diff --git a/utils/timing_stats.h b/utils/timing_stats.h
index 0a9f7656..69a1cf4b 100644
--- a/utils/timing_stats.h
+++ b/utils/timing_stats.h
@@ -1,5 +1,5 @@
-#ifndef _TIMING_STATS_H_
-#define _TIMING_STATS_H_
+#ifndef TIMING_STATS_H_
+#define TIMING_STATS_H_
 
 #include <string>
 #include <map>
diff --git a/utils/verbose.h b/utils/verbose.h
index 73476383..e39e23cb 100644
--- a/utils/verbose.h
+++ b/utils/verbose.h
@@ -1,5 +1,5 @@
-#ifndef _VERBOSE_H_
-#define _VERBOSE_H_
+#ifndef VERBOSE_H_
+#define VERBOSE_H_
 
 extern bool SILENT;
 
diff --git a/utils/weights.h b/utils/weights.h
index 920fdd75..0bd4c2d9 100644
--- a/utils/weights.h
+++ b/utils/weights.h
@@ -1,5 +1,5 @@
-#ifndef _WEIGHTS_H_
-#define _WEIGHTS_H_
+#ifndef WEIGHTS_H_
+#define WEIGHTS_H_
 
 #include <string>
 #include <vector>
diff --git a/utils/wordid.h b/utils/wordid.h
index 714dcd0b..3aa6cc23 100644
--- a/utils/wordid.h
+++ b/utils/wordid.h
@@ -1,5 +1,5 @@
-#ifndef _WORD_ID_H_
-#define _WORD_ID_H_
+#ifndef WORD_ID_H_
+#define WORD_ID_H_
 
 #include <limits>
 
-- 
cgit v1.2.3


From 2931396900c89eb19a50407955574960c364d0ee Mon Sep 17 00:00:00 2001
From: "Wu, Ke" <wuke@cs.umd.edu>
Date: Sun, 12 Oct 2014 16:30:02 -0400
Subject: Cherry picked Mr.MIRA compatibility mode code

---
 decoder/decoder.cc     | 39 ++++++++++++++++++++++++++++-------
 decoder/oracle_bleu.h  | 37 +++++++++++++++++++++++++--------
 utils/Makefile.am      |  3 ++-
 utils/b64featvector.cc | 55 ++++++++++++++++++++++++++++++++++++++++++++++++++
 utils/b64featvector.h  | 12 +++++++++++
 5 files changed, 130 insertions(+), 16 deletions(-)
 create mode 100644 utils/b64featvector.cc
 create mode 100644 utils/b64featvector.h

(limited to 'utils')

diff --git a/decoder/decoder.cc b/decoder/decoder.cc
index c384c33f..93282576 100644
--- a/decoder/decoder.cc
+++ b/decoder/decoder.cc
@@ -17,6 +17,7 @@ namespace std { using std::tr1::unordered_map; }
 #include "fdict.h"
 #include "timing_stats.h"
 #include "verbose.h"
+#include "b64featvector.h"
 
 #include "translator.h"
 #include "phrasebased_translator.h"
@@ -195,7 +196,7 @@ struct DecoderImpl {
       }
       forest.PruneInsideOutside(beam_prune,density_prune,pm,false,1);
       if (!forestname.empty()) forestname=" "+forestname;
-      if (!SILENT) { 
+      if (!SILENT) {
         forest_stats(forest,"  Pruned "+forestname+" forest",false,false);
         cerr << "  Pruned "<<forestname<<" forest portion of edges kept: "<<forest.edges_.size()/presize<<endl;
       }
@@ -261,7 +262,7 @@ struct DecoderImpl {
       assert(ref);
       LatticeTools::ConvertTextOrPLF(sref, ref);
     }
-  } 
+  }
 
   // used to construct the suffix string to get the name of arguments for multiple passes
   // e.g., the "2" in --weights2
@@ -284,7 +285,7 @@ struct DecoderImpl {
   boost::shared_ptr<RandomNumberGenerator<boost::mt19937> > rng;
   int sample_max_trans;
   bool aligner_mode;
-  bool graphviz; 
+  bool graphviz;
   bool joshua_viz;
   bool encode_b64;
   bool kbest;
@@ -301,6 +302,7 @@ struct DecoderImpl {
   bool feature_expectations; // TODO Observer
   bool output_training_vector; // TODO Observer
   bool remove_intersected_rule_annotations;
+  bool mr_mira_compat;  // Mr.MIRA compatibility mode.
   boost::scoped_ptr<IncrementalBase> incremental;
 
 
@@ -414,7 +416,8 @@ DecoderImpl::DecoderImpl(po::variables_map& conf, int argc, char** argv, istream
         ("vector_format",po::value<string>()->default_value("b64"), "Sparse vector serialization format for feature expectations or gradients, includes (text or b64)")
         ("combine_size,C",po::value<int>()->default_value(1), "When option -G is used, process this many sentence pairs before writing the gradient (1=emit after every sentence pair)")
         ("forest_output,O",po::value<string>(),"Directory to write forests to")
-        ("remove_intersected_rule_annotations", "After forced decoding is completed, remove nonterminal annotations (i.e., the source side spans)");
+        ("remove_intersected_rule_annotations", "After forced decoding is completed, remove nonterminal annotations (i.e., the source side spans)")
+        ("mr_mira_compat", "Mr.MIRA compatibility mode (applies weight delta if available; outputs number of lines before k-best)");
 
   // ob.AddOptions(&opts);
   po::options_description clo("Command line options");
@@ -666,6 +669,7 @@ DecoderImpl::DecoderImpl(po::variables_map& conf, int argc, char** argv, istream
   get_oracle_forest = conf.count("get_oracle_forest");
   oracle.show_derivation=conf.count("show_derivations");
   remove_intersected_rule_annotations = conf.count("remove_intersected_rule_annotations");
+  mr_mira_compat = conf.count("mr_mira_compat");
 
   combine_size = conf["combine_size"].as<int>();
   if (combine_size < 1) combine_size = 1;
@@ -699,6 +703,24 @@ void Decoder::AddSupplementalGrammarFromString(const std::string& grammar_string
   static_cast<SCFGTranslator&>(*pimpl_->translator).AddSupplementalGrammarFromString(grammar_string);
 }
 
+static inline void ApplyWeightDelta(const string &delta_b64, vector<weight_t> *weights) {
+  SparseVector<weight_t> delta;
+  DecodeFeatureVector(delta_b64, &delta);
+  if (delta.empty()) return;
+  // Apply updates
+  for (SparseVector<weight_t>::iterator dit = delta.begin();
+       dit != delta.end(); ++dit) {
+    int feat_id = dit->first;
+    union { weight_t weight; unsigned long long repr; } feat_delta;
+    feat_delta.weight = dit->second;
+    if (!SILENT)
+      cerr << "[decoder weight update] " << FD::Convert(feat_id) << " " << feat_delta.weight
+           << " = " << hex << feat_delta.repr << endl;
+    if (weights->size() <= feat_id) weights->resize(feat_id + 1);
+    (*weights)[feat_id] += feat_delta.weight;
+  }
+}
+
 bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
   string buf = input;
   NgramCache::Clear();   // clear ngram cache for remote LM (if used)
@@ -709,6 +731,10 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
   if (sgml.find("id") != sgml.end())
     sent_id = atoi(sgml["id"].c_str());
 
+  // Add delta from input to weights before decoding
+  if (mr_mira_compat)
+    ApplyWeightDelta(sgml["delta"], init_weights.get());
+
   if (!SILENT) {
     cerr << "\nINPUT: ";
     if (buf.size() < 100)
@@ -947,7 +973,7 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
     if (kbest && !has_ref) {
       //TODO: does this work properly?
       const string deriv_fname = conf.count("show_derivations") ? str("show_derivations",conf) : "-";
-      oracle.DumpKBest(sent_id, forest, conf["k_best"].as<int>(), unique_kbest,"-", deriv_fname);
+      oracle.DumpKBest(sent_id, forest, conf["k_best"].as<int>(), unique_kbest,mr_mira_compat, smeta.GetSourceLength(), "-", deriv_fname);
     } else if (csplit_output_plf) {
       cout << HypergraphIO::AsPLF(forest, false) << endl;
     } else {
@@ -1078,7 +1104,7 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
       if (conf.count("graphviz")) forest.PrintGraphviz();
       if (kbest) {
         const string deriv_fname = conf.count("show_derivations") ? str("show_derivations",conf) : "-";
-        oracle.DumpKBest(sent_id, forest, conf["k_best"].as<int>(), unique_kbest,"-", deriv_fname);
+        oracle.DumpKBest(sent_id, forest, conf["k_best"].as<int>(), unique_kbest, mr_mira_compat, smeta.GetSourceLength(), "-", deriv_fname);
       }
       if (conf.count("show_conditional_prob")) {
         const prob_t ref_z = Inside<prob_t, EdgeProb>(forest);
@@ -1098,4 +1124,3 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
   o->NotifyDecodingComplete(smeta);
   return true;
 }
-
diff --git a/decoder/oracle_bleu.h b/decoder/oracle_bleu.h
index d2c4715c..75db61e8 100644
--- a/decoder/oracle_bleu.h
+++ b/decoder/oracle_bleu.h
@@ -21,6 +21,7 @@
 #include "kbest.h"
 #include "timing_stats.h"
 #include "sentences.h"
+#include "b64featvector.h"
 
 //TODO: put function impls into .cc
 //TODO: move Translation into its own .h and use in cdec
@@ -253,18 +254,28 @@ struct OracleBleu {
 
   bool show_derivation;
   template <class Filter>
-  void kbest(int sent_id,Hypergraph const& forest,int k,std::ostream &kbest_out=std::cout,std::ostream &deriv_out=std::cerr) {
+  void kbest(int sent_id, Hypergraph const& forest, int k, bool mr_mira_compat,
+             int src_len, std::ostream& kbest_out = std::cout,
+             std::ostream& deriv_out = std::cerr) {
     using namespace std;
     using namespace boost;
     typedef KBest::KBestDerivations<Sentence, ESentenceTraversal,Filter> K;
     K kbest(forest,k);
     //add length (f side) src length of this sentence to the psuedo-doc src length count
     float curr_src_length = doc_src_length + tmp_src_length;
-    for (int i = 0; i < k; ++i) {
+    if (mr_mira_compat) kbest_out << k << "\n";
+    int i = 0;
+    for (; i < k; ++i) {
       typename K::Derivation *d = kbest.LazyKthBest(forest.nodes_.size() - 1, i);
       if (!d) break;
-      kbest_out << sent_id << " ||| " << TD::GetString(d->yield) << " ||| "
-                << d->feature_values << " ||| " << log(d->score);
+      kbest_out << sent_id << " ||| ";
+      if (mr_mira_compat) kbest_out << src_len << " ||| ";
+      kbest_out << TD::GetString(d->yield) << " ||| ";
+      if (mr_mira_compat)
+        kbest_out << EncodeFeatureVector(d->feature_values);
+      else
+        kbest_out << d->feature_values;
+      kbest_out << " ||| " << log(d->score);
       if (!refs.empty()) {
         ScoreP sentscore = GetScore(d->yield,sent_id);
         sentscore->PlusEquals(*doc_score,float(1));
@@ -279,10 +290,17 @@ struct OracleBleu {
         deriv_out<<"\n"<<flush;
       }
     }
+    if (mr_mira_compat) {
+      for (; i < k; ++i) kbest_out << "\n";
+      kbest_out << flush;
+    }
   }
 
 // TODO decoder output should probably be moved to another file - how about oracle_bleu.h
-  void DumpKBest(const int sent_id, const Hypergraph& forest, const int k, const bool unique, std::string const &kbest_out_filename_, std::string const &deriv_out_filename_) {
+  void DumpKBest(const int sent_id, const Hypergraph& forest, const int k,
+                 const bool unique, const bool mr_mira_compat,
+                 const int src_len, std::string const& kbest_out_filename_,
+                 std::string const& deriv_out_filename_) {
 
     WriteFile ko(kbest_out_filename_);
     std::cerr << "Output kbest to " << kbest_out_filename_ <<std::endl;
@@ -295,9 +313,11 @@ struct OracleBleu {
     WriteFile oderiv(sderiv.str());
 
     if (!unique)
-      kbest<KBest::NoFilter<std::vector<WordID> > >(sent_id,forest,k,ko.get(),oderiv.get());
+      kbest<KBest::NoFilter<std::vector<WordID> > >(
+          sent_id, forest, k, mr_mira_compat, src_len, ko.get(), oderiv.get());
     else {
-      kbest<KBest::FilterUnique>(sent_id,forest,k,ko.get(),oderiv.get());
+      kbest<KBest::FilterUnique>(sent_id, forest, k, mr_mira_compat, src_len,
+                                 ko.get(), oderiv.get());
     }
   }
 
@@ -305,7 +325,8 @@ void DumpKBest(std::string const& suffix,const int sent_id, const Hypergraph& fo
   {
     std::ostringstream kbest_string_stream;
     kbest_string_stream << forest_output << "/kbest_"<<suffix<< "." << sent_id;
-    DumpKBest(sent_id, forest, k, unique, kbest_string_stream.str(), "-");
+    DumpKBest(sent_id, forest, k, unique, false, -1, kbest_string_stream.str(),
+              "-");
   }
 
 };
diff --git a/utils/Makefile.am b/utils/Makefile.am
index 727fa8a5..64f6d433 100644
--- a/utils/Makefile.am
+++ b/utils/Makefile.am
@@ -22,6 +22,7 @@ libutils_a_SOURCES = \
   alias_sampler.h \
   alignment_io.h \
   array2d.h \
+  b64featvector.h \
   b64tools.h \
   batched_append.h \
   city.h \
@@ -70,6 +71,7 @@ libutils_a_SOURCES = \
   fast_lexical_cast.hpp \
   intrusive_refcount.hpp \
   alignment_io.cc \
+  b64featvector.cc \
   b64tools.cc \
   corpus_tools.cc \
   dict.cc \
@@ -117,4 +119,3 @@ stringlib_test_LDADD = libutils.a $(BOOST_UNIT_TEST_FRAMEWORK_LDFLAGS) $(BOOST_U
 # do NOT NOT NOT add any other -I includes NO NO NO NO NO ######
 AM_CPPFLAGS = -DBOOST_TEST_DYN_LINK -W -Wall -I. -I$(top_srcdir) -DTEST_DATA=\"$(top_srcdir)/utils/test_data\"
 ################################################################
-
diff --git a/utils/b64featvector.cc b/utils/b64featvector.cc
new file mode 100644
index 00000000..c7d08b29
--- /dev/null
+++ b/utils/b64featvector.cc
@@ -0,0 +1,55 @@
+#include "b64featvector.h"
+
+#include <sstream>
+#include <boost/scoped_array.hpp>
+#include "b64tools.h"
+#include "fdict.h"
+
+using namespace std;
+
+static inline void EncodeFeatureWeight(const string &featname, weight_t weight,
+                                       ostream *output) {
+  output->write(featname.data(), featname.size() + 1);
+  output->write(reinterpret_cast<char *>(&weight), sizeof(weight_t));
+}
+
+string EncodeFeatureVector(const SparseVector<weight_t> &vec) {
+  string b64;
+  {
+    ostringstream base64_strm;
+    {
+      ostringstream strm;
+      for (SparseVector<weight_t>::const_iterator it = vec.begin();
+           it != vec.end(); ++it)
+        if (it->second != 0)
+          EncodeFeatureWeight(FD::Convert(it->first), it->second, &strm);
+      string data(strm.str());
+      B64::b64encode(data.data(), data.size(), &base64_strm);
+    }
+    b64 = base64_strm.str();
+  }
+  return b64;
+}
+
+void DecodeFeatureVector(const string &data, SparseVector<weight_t> *vec) {
+  vec->clear();
+  if (data.empty()) return;
+  // Decode data
+  size_t b64_len = data.size(), len = b64_len / 4 * 3;
+  boost::scoped_array<char> buf(new char[len]);
+  bool res =
+      B64::b64decode(reinterpret_cast<const unsigned char *>(data.data()),
+                     b64_len, buf.get(), len);
+  assert(res);
+  // Apply updates
+  size_t cur = 0;
+  while (cur < len) {
+    string feat_name(buf.get() + cur);
+    if (feat_name.empty()) break;  // Encountered trailing \0
+    int feat_id = FD::Convert(feat_name);
+    weight_t feat_delta =
+        *reinterpret_cast<weight_t *>(buf.get() + cur + feat_name.size() + 1);
+    (*vec)[feat_id] = feat_delta;
+    cur += feat_name.size() + 1 + sizeof(weight_t);
+  }
+}
diff --git a/utils/b64featvector.h b/utils/b64featvector.h
new file mode 100644
index 00000000..6ac04d44
--- /dev/null
+++ b/utils/b64featvector.h
@@ -0,0 +1,12 @@
+#ifndef _B64FEATVECTOR_H_
+#define _B64FEATVECTOR_H_
+
+#include <string>
+
+#include "sparse_vector.h"
+#include "weights.h"
+
+std::string EncodeFeatureVector(const SparseVector<weight_t> &);
+void DecodeFeatureVector(const std::string &, SparseVector<weight_t> *);
+
+#endif  // _B64FEATVECTOR_H_
-- 
cgit v1.2.3


From 2e485c55fb17b75c0b153af349d8283ab0e9384f Mon Sep 17 00:00:00 2001
From: Chris Dyer <cdyer@allegro.clab.cs.cmu.edu>
Date: Sat, 18 Oct 2014 21:46:49 -0400
Subject: test sparse vector serialization

---
 utils/sv_test.cc | 31 +++++++++++++++++++++++++++++++
 1 file changed, 31 insertions(+)

(limited to 'utils')

diff --git a/utils/sv_test.cc b/utils/sv_test.cc
index 67df8c57..b006e66d 100644
--- a/utils/sv_test.cc
+++ b/utils/sv_test.cc
@@ -1,7 +1,12 @@
 #define BOOST_TEST_MODULE WeightsTest
 #include <boost/test/unit_test.hpp>
 #include <boost/test/floating_point_comparison.hpp>
+#include <boost/archive/text_oarchive.hpp>
+#include <boost/archive/text_iarchive.hpp>
+#include <sstream>
+#include <string>
 #include "sparse_vector.h"
+#include "fdict.h"
 
 using namespace std;
 
@@ -33,3 +38,29 @@ BOOST_AUTO_TEST_CASE(Division) {
   x /= -1;
   BOOST_CHECK(x == y);
 }
+
+BOOST_AUTO_TEST_CASE(Serialization) {
+  string arc;
+  FD::dict_.clear();
+  {
+    SparseVector<double> x;
+    x.set_value(FD::Convert("Feature1"), 1.0);
+    x.set_value(FD::Convert("Pi"), 3.14);
+    ostringstream os;
+    boost::archive::text_oarchive oa(os);
+    oa << x;
+    arc = os.str();
+  }
+  FD::dict_.clear();
+  FD::Convert("SomeNewString");
+  {
+    SparseVector<double> x;
+    istringstream is(arc);
+    boost::archive::text_iarchive ia(is);
+    ia >> x;
+    cerr << x << endl;
+    BOOST_CHECK_CLOSE(x.get(FD::Convert("Pi")), 3.14, 1e-9);
+    BOOST_CHECK_CLOSE(x.get(FD::Convert("Feature1")), 1.0, 1e-9);
+  }
+}
+
-- 
cgit v1.2.3


From 011a87cfe6d9cc702cb4a8a6d9a765556e460af9 Mon Sep 17 00:00:00 2001
From: Chris Dyer <cdyer@allegro.clab.cs.cmu.edu>
Date: Sun, 19 Oct 2014 05:24:21 -0400
Subject: stop switch to boost serialization for hypergraph IO

---
 decoder/decoder.cc                                 |   4 +-
 decoder/forest_writer.cc                           |   4 +-
 decoder/hg.h                                       |  52 +++++++++++
 decoder/hg_io.cc                                   | 101 +++------------------
 decoder/hg_io.h                                    |   5 +-
 decoder/hg_test.cc                                 |  39 +++++---
 decoder/rule_lexer.ll                              |   1 +
 python/cdec/hypergraph.pxd                         |   3 +-
 training/dpmert/mr_dpmert_generate_mapper_input.cc |   2 +-
 training/dpmert/mr_dpmert_map.cc                   |   4 +-
 training/minrisk/minrisk_optimize.cc               |   2 +-
 training/pro/mr_pro_map.cc                         |   2 +-
 training/rampion/rampion_cccp.cc                   |   2 +-
 training/utils/grammar_convert.cc                  |   5 +-
 utils/small_vector.h                               |  16 ++++
 utils/small_vector_test.cc                         |  30 ++++++
 16 files changed, 156 insertions(+), 116 deletions(-)

(limited to 'utils')

diff --git a/decoder/decoder.cc b/decoder/decoder.cc
index 3cc77d27..f8214f5f 100644
--- a/decoder/decoder.cc
+++ b/decoder/decoder.cc
@@ -930,7 +930,7 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
       Hypergraph new_hg;
       {
         ReadFile rf(writer.fname_);
-        bool succeeded = HypergraphIO::ReadFromJSON(rf.stream(), &new_hg);
+        bool succeeded = HypergraphIO::ReadFromBinary(rf.stream(), &new_hg);
         if (!succeeded) abort();
       }
       HG::Union(forest, &new_hg);
@@ -1023,7 +1023,7 @@ bool DecoderImpl::Decode(const string& input, DecoderObserver* o) {
           Hypergraph new_hg;
           {
             ReadFile rf(writer.fname_);
-            bool succeeded = HypergraphIO::ReadFromJSON(rf.stream(), &new_hg);
+            bool succeeded = HypergraphIO::ReadFromBinary(rf.stream(), &new_hg);
             if (!succeeded) abort();
           }
           HG::Union(forest, &new_hg);
diff --git a/decoder/forest_writer.cc b/decoder/forest_writer.cc
index 6e4cccb3..c072e599 100644
--- a/decoder/forest_writer.cc
+++ b/decoder/forest_writer.cc
@@ -11,13 +11,13 @@
 using namespace std;
 
 ForestWriter::ForestWriter(const std::string& path, int num) :
-  fname_(path + '/' + boost::lexical_cast<string>(num) + ".json.gz"), used_(false) {}
+  fname_(path + '/' + boost::lexical_cast<string>(num) + ".bin.gz"), used_(false) {}
 
 bool ForestWriter::Write(const Hypergraph& forest, bool minimal_rules) {
   assert(!used_);
   used_ = true;
   cerr << "  Writing forest to " << fname_ << endl;
   WriteFile wf(fname_);
-  return HypergraphIO::WriteToJSON(forest, minimal_rules, wf.stream());
+  return HypergraphIO::WriteToBinary(forest, wf.stream());
 }
 
diff --git a/decoder/hg.h b/decoder/hg.h
index 256f650f..124eab86 100644
--- a/decoder/hg.h
+++ b/decoder/hg.h
@@ -18,6 +18,7 @@
 #include <string>
 #include <vector>
 #include <boost/shared_ptr.hpp>
+#include <boost/serialization/vector.hpp>
 
 #include "feature_vector.h"
 #include "small_vector.h"
@@ -69,6 +70,18 @@ namespace HG {
     short int j_;
     short int prev_i_;
     short int prev_j_;
+    template<class Archive>
+    void serialize(Archive & ar, const unsigned int version) {
+      ar & head_node_;
+      ar & tail_nodes_;
+      ar & rule_;
+      ar & feature_values_;
+      ar & i_;
+      ar & j_;
+      ar & prev_i_;
+      ar & prev_j_;
+      ar & id_;
+    }
     void show(std::ostream &o,unsigned mask=SPAN|RULE) const {
       o<<'{';
       if (mask&CATEGORY)
@@ -149,6 +162,24 @@ namespace HG {
     WordID NT() const { return -cat_; }
     EdgesVector in_edges_;   // an in edge is an edge with this node as its head.  (in edges come from the bottom up to us)  indices in edges_
     EdgesVector out_edges_;  // an out edge is an edge with this node as its tail.  (out edges leave us up toward the top/goal). indices in edges_
+    template<class Archive>
+    void save(Archive & ar, const unsigned int version) const {
+      ar & node_hash;
+      ar & id_;
+      ar & TD::Convert(-cat_);
+      ar & in_edges_;
+      ar & out_edges_;
+    }
+    template<class Archive>
+    void load(Archive & ar, const unsigned int version) {
+      ar & node_hash;
+      ar & id_;
+      std::string cat; ar & cat;
+      cat_ = -TD::Convert(cat);
+      ar & in_edges_;
+      ar & out_edges_;
+    }
+    BOOST_SERIALIZATION_SPLIT_MEMBER()
     void copy_fixed(Node const& o) { // nonstructural fields only - structural ones are managed by sorting/pruning/subsetting
       node_hash = o.node_hash;
       cat_=o.cat_;
@@ -492,6 +523,27 @@ public:
   void set_ids(); // resync edge,node .id_
   void check_ids() const; // assert that .id_ have been kept in sync
 
+  template<class Archive>
+  void save(Archive & ar, const unsigned int version) const {
+    unsigned ns = nodes_.size(); ar & ns;
+    unsigned es = edges_.size(); ar & es;
+    for (auto& n : nodes_) ar & n;
+    for (auto& e : edges_) ar & e;
+    int x;
+    x = edges_topo_; ar & x;
+    x = is_linear_chain_; ar & x;
+  }
+  template<class Archive>
+  void load(Archive & ar, const unsigned int version) {
+    unsigned ns; ar & ns; nodes_.resize(ns);
+    unsigned es; ar & es; edges_.resize(es);
+    for (auto& n : nodes_) ar & n;
+    for (auto& e : edges_) ar & e;
+    int x;
+    ar & x; edges_topo_ = x;
+    ar & x; is_linear_chain_ = x;
+  }
+  BOOST_SERIALIZATION_SPLIT_MEMBER()
 private:
   Hypergraph(int num_nodes, int num_edges, bool is_lc) : is_linear_chain_(is_lc), nodes_(num_nodes), edges_(num_edges),edges_topo_(true) {}
 };
diff --git a/decoder/hg_io.cc b/decoder/hg_io.cc
index eb0be3d4..67760fb1 100644
--- a/decoder/hg_io.cc
+++ b/decoder/hg_io.cc
@@ -6,6 +6,10 @@
 #include <sstream>
 #include <iostream>
 
+#include <boost/archive/binary_iarchive.hpp>
+#include <boost/archive/binary_oarchive.hpp>
+#include <boost/serialization/shared_ptr.hpp>
+
 #include "fast_lexical_cast.hpp"
 
 #include "tdict.h"
@@ -271,97 +275,16 @@ bool HypergraphIO::ReadFromJSON(istream* in, Hypergraph* hg) {
   return reader.Parse(in);
 }
 
-static void WriteRule(const TRule& r, ostream* out) {
-  if (!r.lhs_) { (*out) << "[X] ||| "; }
-  JSONParser::WriteEscapedString(r.AsString(), out);
+bool HypergraphIO::ReadFromBinary(istream* in, Hypergraph* hg) {
+  boost::archive::binary_iarchive oa(*in);
+  hg->clear();
+  oa >> *hg;
+  return true;
 }
 
-bool HypergraphIO::WriteToJSON(const Hypergraph& hg, bool remove_rules, ostream* out) {
-  if (hg.empty()) { *out << "{}\n"; return true; }
-  map<const TRule*, int> rid;
-  ostream& o = *out;
-  rid[NULL] = 0;
-  o << '{';
-  if (!remove_rules) {
-    o << "\"rules\":[";
-    for (int i = 0; i < hg.edges_.size(); ++i) {
-      const TRule* r = hg.edges_[i].rule_.get();
-      int &id = rid[r];
-      if (!id) {
-        id=rid.size() - 1;
-        if (id > 1) o << ',';
-        o << id << ',';
-        WriteRule(*r, &o);
-      };
-    }
-    o << "],";
-  }
-  const bool use_fdict = FD::NumFeats() < 1000;
-  if (use_fdict) {
-    o << "\"features\":[";
-    for (int i = 1; i < FD::NumFeats(); ++i) {
-      o << (i==1 ? "":",");
-      JSONParser::WriteEscapedString(FD::Convert(i), &o);
-    }
-    o << "],";
-  }
-  vector<int> edgemap(hg.edges_.size(), -1);  // edges may be in non-topo order
-  int edge_count = 0;
-  for (int i = 0; i < hg.nodes_.size(); ++i) {
-    const Hypergraph::Node& node = hg.nodes_[i];
-    if (i > 0) { o << ","; }
-    o << "\"edges\":[";
-    for (int j = 0; j < node.in_edges_.size(); ++j) {
-      const Hypergraph::Edge& edge = hg.edges_[node.in_edges_[j]];
-      edgemap[edge.id_] = edge_count;
-      ++edge_count;
-      o << (j == 0 ? "" : ",") << "{";
-
-      o << "\"tail\":[";
-      for (int k = 0; k < edge.tail_nodes_.size(); ++k) {
-        o << (k > 0 ? "," : "") << edge.tail_nodes_[k];
-      }
-      o << "],";
-
-      o << "\"spans\":[" << edge.i_ << "," << edge.j_ << "," << edge.prev_i_ << "," << edge.prev_j_ << "],";
-
-      o << "\"feats\":[";
-      bool first = true;
-      for (SparseVector<double>::const_iterator it = edge.feature_values_.begin(); it != edge.feature_values_.end(); ++it) {
-        if (!it->second) continue;   // don't write features that have a zero value
-        if (!it->first) continue;    // if the feature set was frozen this might happen
-        if (!first) o << ',';
-        if (use_fdict)
-          o << (it->first - 1);
-        else {
-	  JSONParser::WriteEscapedString(FD::Convert(it->first), &o);
-        }
-	o << ',' << it->second;
-        first = false;
-      }
-      o << "]";
-      if (!remove_rules) { o << ",\"rule\":" << rid[edge.rule_.get()]; }
-      o << "}";
-    }
-    o << "],";
-
-    o << "\"node\":{\"in_edges\":[";
-    for (int j = 0; j < node.in_edges_.size(); ++j) {
-      int mapped_edge = edgemap[node.in_edges_[j]];
-      assert(mapped_edge >= 0);
-      o << (j == 0 ? "" : ",") << mapped_edge;
-    }
-    o << "]";
-    if (node.cat_ < 0) {
-       o << ",\"cat\":";
-       JSONParser::WriteEscapedString(TD::Convert(node.cat_ * -1), &o);
-    }
-    char buf[48];
-    sprintf(buf, "%016lX", node.node_hash);
-    o << ",\"node_hash\":\"" << buf << "\"";
-    o << "}";
-  }
-  o << "}\n";
+bool HypergraphIO::WriteToBinary(const Hypergraph& hg, ostream* out) {
+  boost::archive::binary_oarchive oa(*out);
+  oa << hg;
   return true;
 }
 
diff --git a/decoder/hg_io.h b/decoder/hg_io.h
index 5a2bd808..5ba86f69 100644
--- a/decoder/hg_io.h
+++ b/decoder/hg_io.h
@@ -18,10 +18,11 @@ struct HypergraphIO {
   // see test_data/small.json.gz for an email encoding
   static bool ReadFromJSON(std::istream* in, Hypergraph* out);
 
+  static bool ReadFromBinary(std::istream* in, Hypergraph* out);
+  static bool WriteToBinary(const Hypergraph& hg, std::ostream* out);
+
   // if remove_rules is used, the hypergraph is serialized without rule information
   // (so it only contains structure and feature information)
-  static bool WriteToJSON(const Hypergraph& hg, bool remove_rules, std::ostream* out);
-
   static void WriteAsCFG(const Hypergraph& hg);
 
   // Write only the target size information in bottom-up order.  
diff --git a/decoder/hg_test.cc b/decoder/hg_test.cc
index 5cb8626a..25eddcec 100644
--- a/decoder/hg_test.cc
+++ b/decoder/hg_test.cc
@@ -1,6 +1,11 @@
 #define BOOST_TEST_MODULE hg_test
 #include <boost/test/unit_test.hpp>
 #include <boost/test/floating_point_comparison.hpp>
+#include <boost/archive/text_oarchive.hpp>
+#include <boost/archive/text_iarchive.hpp>
+#include <boost/serialization/shared_ptr.hpp>
+#include <boost/serialization/vector.hpp>
+#include <sstream>
 #include <iostream>
 #include "tdict.h"
 
@@ -427,19 +432,29 @@ BOOST_AUTO_TEST_CASE(TestGenericKBest) {
   }
 }
 
-BOOST_AUTO_TEST_CASE(TestReadWriteHG) {
+BOOST_AUTO_TEST_CASE(TestReadWriteHG_Boost) {
   std::string path(boost::unit_test::framework::master_test_suite().argc == 2 ? boost::unit_test::framework::master_test_suite().argv[1] : TEST_DATA);
-  Hypergraph hg,hg2;
-  CreateHG(path, &hg);
-  hg.edges_.front().j_ = 23;
-  hg.edges_.back().prev_i_ = 99;
-  ostringstream os;
-  HypergraphIO::WriteToJSON(hg, false, &os);
-  istringstream is(os.str());
-  HypergraphIO::ReadFromJSON(&is, &hg2);
-  BOOST_CHECK_EQUAL(hg2.NumberOfPaths(), hg.NumberOfPaths());
-  BOOST_CHECK_EQUAL(hg2.edges_.front().j_, 23);
-  BOOST_CHECK_EQUAL(hg2.edges_.back().prev_i_, 99);
+  Hypergraph hg;
+  Hypergraph hg2;
+  std::string out;
+  {
+    CreateHG(path, &hg);
+    hg.edges_.front().j_ = 23;
+    hg.edges_.back().prev_i_ = 99;
+    ostringstream os;
+    boost::archive::text_oarchive oa(os);
+    oa << hg;
+    out = os.str();
+  }
+  {
+    cerr << out << endl;
+    istringstream is(out);
+    boost::archive::text_iarchive ia(is);
+    ia >> hg2;
+    BOOST_CHECK_EQUAL(hg2.NumberOfPaths(), hg.NumberOfPaths());
+    BOOST_CHECK_EQUAL(hg2.edges_.front().j_, 23);
+    BOOST_CHECK_EQUAL(hg2.edges_.back().prev_i_, 99);
+  }
 }
 
 BOOST_AUTO_TEST_SUITE_END()
diff --git a/decoder/rule_lexer.ll b/decoder/rule_lexer.ll
index d4a8d86b..8b48ab7b 100644
--- a/decoder/rule_lexer.ll
+++ b/decoder/rule_lexer.ll
@@ -356,6 +356,7 @@ void RuleLexer::ReadRules(std::istream* in, RuleLexer::RuleCallback func, const
 
 void RuleLexer::ReadRule(const std::string& srule, RuleCallback func, bool mono, void* extra) {
   init_default_feature_names();
+  scfglex_fname = srule;
   lex_mono_rules = mono;
   lex_line = 1;
   rule_callback_extra = extra;
diff --git a/python/cdec/hypergraph.pxd b/python/cdec/hypergraph.pxd
index 1e150bbc..9780cf8b 100644
--- a/python/cdec/hypergraph.pxd
+++ b/python/cdec/hypergraph.pxd
@@ -63,7 +63,8 @@ cdef extern from "decoder/viterbi.h":
 cdef extern from "decoder/hg_io.h" namespace "HypergraphIO":
     # Hypergraph JSON I/O
     bint ReadFromJSON(istream* inp, Hypergraph* out)
-    bint WriteToJSON(Hypergraph& hg, bint remove_rules, ostream* out)
+    bint ReadFromBinary(istream* inp, Hypergraph* out)
+    bint WriteToBinary(Hypergraph& hg, ostream* out)
     # Hypergraph PLF I/O
     void ReadFromPLF(string& inp, Hypergraph* out)
     string AsPLF(Hypergraph& hg, bint include_global_parentheses)
diff --git a/training/dpmert/mr_dpmert_generate_mapper_input.cc b/training/dpmert/mr_dpmert_generate_mapper_input.cc
index 199cd23a..3fa2f476 100644
--- a/training/dpmert/mr_dpmert_generate_mapper_input.cc
+++ b/training/dpmert/mr_dpmert_generate_mapper_input.cc
@@ -70,7 +70,7 @@ int main(int argc, char** argv) {
   unsigned dev_set_size = conf["dev_set_size"].as<unsigned>();
   for (unsigned i = 0; i < dev_set_size; ++i) {
     for (unsigned j = 0; j < directions.size(); ++j) {
-      cout << forest_repository << '/' << i << ".json.gz " << i << ' ';
+      cout << forest_repository << '/' << i << ".bin.gz " << i << ' ';
       print(cout, origin, "=", ";");
       cout << ' ';
       print(cout, directions[j], "=", ";");
diff --git a/training/dpmert/mr_dpmert_map.cc b/training/dpmert/mr_dpmert_map.cc
index d1efcf96..2bf3f8fc 100644
--- a/training/dpmert/mr_dpmert_map.cc
+++ b/training/dpmert/mr_dpmert_map.cc
@@ -83,7 +83,7 @@ int main(int argc, char** argv) {
     istringstream is(line);
     int sent_id;
     string file, s_origin, s_direction;
-    // path-to-file (JSON) sent_ed starting-point search-direction
+    // path-to-file sent_ed starting-point search-direction
     is >> file >> sent_id >> s_origin >> s_direction;
     SparseVector<double> origin;
     ReadSparseVectorString(s_origin, &origin);
@@ -93,7 +93,7 @@ int main(int argc, char** argv) {
     if (last_file != file) {
       last_file = file;
       ReadFile rf(file);
-      HypergraphIO::ReadFromJSON(rf.stream(), &hg);
+      HypergraphIO::ReadFromBinary(rf.stream(), &hg);
     }
     const ConvexHullWeightFunction wf(origin, direction);
     const ConvexHull hull = Inside<ConvexHull, ConvexHullWeightFunction>(hg, NULL, wf);
diff --git a/training/minrisk/minrisk_optimize.cc b/training/minrisk/minrisk_optimize.cc
index da8b5260..a2938fb0 100644
--- a/training/minrisk/minrisk_optimize.cc
+++ b/training/minrisk/minrisk_optimize.cc
@@ -178,7 +178,7 @@ int main(int argc, char** argv) {
     ReadFile rf(file);
     if (kis.size() % 5 == 0) { cerr << '.'; }
     if (kis.size() % 200 == 0) { cerr << " [" << kis.size() << "]\n"; }
-    HypergraphIO::ReadFromJSON(rf.stream(), &hg);
+    HypergraphIO::ReadFromBinary(rf.stream(), &hg);
     hg.Reweight(weights);
     curkbest.AddKBestCandidates(hg, kbest_size, ds[sent_id]);
     if (kbest_file.size())
diff --git a/training/pro/mr_pro_map.cc b/training/pro/mr_pro_map.cc
index da58cd24..b142fd05 100644
--- a/training/pro/mr_pro_map.cc
+++ b/training/pro/mr_pro_map.cc
@@ -203,7 +203,7 @@ int main(int argc, char** argv) {
     const string kbest_file = os.str();
     if (FileExists(kbest_file))
       J_i.ReadFromFile(kbest_file);
-    HypergraphIO::ReadFromJSON(rf.stream(), &hg);
+    HypergraphIO::ReadFromBinary(rf.stream(), &hg);
     hg.Reweight(weights);
     J_i.AddKBestCandidates(hg, kbest_size, ds[sent_id]);
     J_i.WriteToFile(kbest_file);
diff --git a/training/rampion/rampion_cccp.cc b/training/rampion/rampion_cccp.cc
index 1e36dc51..1c45bac5 100644
--- a/training/rampion/rampion_cccp.cc
+++ b/training/rampion/rampion_cccp.cc
@@ -136,7 +136,7 @@ int main(int argc, char** argv) {
     ReadFile rf(file);
     if (kis.size() % 5 == 0) { cerr << '.'; }
     if (kis.size() % 200 == 0) { cerr << " [" << kis.size() << "]\n"; }
-    HypergraphIO::ReadFromJSON(rf.stream(), &hg);
+    HypergraphIO::ReadFromBinary(rf.stream(), &hg);
     hg.Reweight(weights);
     curkbest.AddKBestCandidates(hg, kbest_size, ds[sent_id]);
     if (kbest_file.size())
diff --git a/training/utils/grammar_convert.cc b/training/utils/grammar_convert.cc
index 5c1b4d4a..000f2a26 100644
--- a/training/utils/grammar_convert.cc
+++ b/training/utils/grammar_convert.cc
@@ -43,7 +43,7 @@ void InitCommandLine(int argc, char** argv, po::variables_map* conf) {
   po::notify(*conf);
 
   if (conf->count("help") || conf->count("input") == 0) {
-    cerr << "\nUsage: grammar_convert [-options]\n\nConverts a grammar file (in Hiero format) into JSON hypergraph.\n";
+    cerr << "\nUsage: grammar_convert [-options]\n\nConverts a grammar file (in Hiero format) into serialized hypergraph.\n";
     cerr << dcmdline_options << endl;
     exit(1);
   }
@@ -254,7 +254,8 @@ void ProcessHypergraph(const vector<double>& w, const po::variables_map& conf, c
   if (w.size() > 0) { hg->Reweight(w); }
   if (conf.count("collapse_weights")) CollapseWeights(hg);
   if (conf["output"].as<string>() == "json") {
-    HypergraphIO::WriteToJSON(*hg, false, &cout);
+    cerr << "NOT IMPLEMENTED ... talk to cdyer if you need this functionality\n";
+    // HypergraphIO::WriteToBinary(*hg, &cout);
     if (!ref.empty()) { cerr << "REF: " << ref << endl; }
   } else {
     vector<WordID> onebest;
diff --git a/utils/small_vector.h b/utils/small_vector.h
index c8cbcb2c..f16bc898 100644
--- a/utils/small_vector.h
+++ b/utils/small_vector.h
@@ -15,6 +15,7 @@
 #include <new>
 #include <stdint.h>
 #include <boost/functional/hash.hpp>
+#include <boost/serialization/map.hpp>
 
 //sizeof(T)/sizeof(T*)>1?sizeof(T)/sizeof(T*):1
 
@@ -297,6 +298,21 @@ public:
     return hash_range(data_.ptr,data_.ptr+size_);
   }
 
+  template<class Archive>
+  void save(Archive & ar, const unsigned int) const {
+    ar & size_;
+    for (unsigned i = 0; i < size_; ++i)
+      ar & (*this)[i];
+  }
+  template<class Archive>
+  void load(Archive & ar, const unsigned int) {
+    uint16_t s;
+    ar & s;
+    this->resize(s);
+    for (unsigned i = 0; i < size_; ++i)
+      ar & (*this)[i];
+  }
+  BOOST_SERIALIZATION_SPLIT_MEMBER()
  private:
   union StorageType {
     T vals[SV_MAX];
diff --git a/utils/small_vector_test.cc b/utils/small_vector_test.cc
index a4eb89ae..9e1a148d 100644
--- a/utils/small_vector_test.cc
+++ b/utils/small_vector_test.cc
@@ -3,6 +3,10 @@
 #define BOOST_TEST_MODULE svTest
 #include <boost/test/unit_test.hpp>
 #include <boost/test/floating_point_comparison.hpp>
+#include <boost/archive/text_oarchive.hpp>
+#include <boost/archive/text_iarchive.hpp>
+#include <string>
+#include <sstream>
 #include <iostream>
 #include <vector>
 
@@ -128,3 +132,29 @@ BOOST_AUTO_TEST_CASE(Small) {
   cerr << sizeof(SmallVectorInt) << endl;
   cerr << sizeof(vector<int>) << endl;
 }
+
+BOOST_AUTO_TEST_CASE(Serialize) {
+  std::string in;
+  {
+    SmallVectorInt v;
+    v.push_back(0);
+    v.push_back(1);
+    v.push_back(-2);
+    ostringstream os;
+    boost::archive::text_oarchive oa(os);
+    oa << v;
+    in = os.str();
+    cerr << in;
+  }
+  {
+    istringstream is(in);
+    boost::archive::text_iarchive ia(is);
+    SmallVectorInt v;
+    ia >> v;
+    BOOST_CHECK_EQUAL(v.size(), 3);
+    BOOST_CHECK_EQUAL(v[0], 0);
+    BOOST_CHECK_EQUAL(v[1], 1);
+    BOOST_CHECK_EQUAL(v[2], -2);
+  }
+}
+
-- 
cgit v1.2.3