diff options
| author | Paul Baltescu <pauldb89@gmail.com> | 2013-04-24 17:18:10 +0100 | 
|---|---|---|
| committer | Paul Baltescu <pauldb89@gmail.com> | 2013-04-24 17:18:10 +0100 | 
| commit | e8b412577b9d3fe2090b9f48443f919cd268c809 (patch) | |
| tree | b46a7b51d365519dfb5170d71bac33be6d3e29b9 /klm/lm/builder/corpus_count.hh | |
| parent | d189426a7ea56b71eb6e25ed02a7b0993cfb56a8 (diff) | |
| parent | 5aee54869aa19cfe9be965e67a472e94449d16da (diff) | |
Merge branch 'master' of https://github.com/redpony/cdec
Diffstat (limited to 'klm/lm/builder/corpus_count.hh')
| -rw-r--r-- | klm/lm/builder/corpus_count.hh | 5 | 
1 files changed, 5 insertions, 0 deletions
| diff --git a/klm/lm/builder/corpus_count.hh b/klm/lm/builder/corpus_count.hh index e255bad1..aa0ed8ed 100644 --- a/klm/lm/builder/corpus_count.hh +++ b/klm/lm/builder/corpus_count.hh @@ -23,6 +23,11 @@ class CorpusCount {      // Memory usage will be DedupeMultipler(order) * block_size + total_chain_size + unknown vocab_hash_size      static float DedupeMultiplier(std::size_t order); +    // How much memory vocabulary will use based on estimated size of the vocab. +    static std::size_t VocabUsage(std::size_t vocab_estimate); + +    // token_count: out. +    // type_count aka vocabulary size.  Initialize to an estimate.  It is set to the exact value.      CorpusCount(util::FilePiece &from, int vocab_write, uint64_t &token_count, WordIndex &type_count, std::size_t entries_per_block);      void Run(const util::stream::ChainPosition &position); | 
