Loading...
Searching...
No Matches
language_dataset.h
Go to the documentation of this file.
28 LanguageDataset(const filesystem::path&, Index maximum_vocabulary_size, Index minimum_token_frequency);
38 const unordered_map<string, Index>& get_input_vocabulary_map() const { return input_vocabulary_map; }
39 const unordered_map<string, Index>& get_target_vocabulary_map() const { return target_vocabulary_map; }
40 const unordered_map<Index, string>& get_target_inverse_vocabulary_map() const { return target_inverse_vocabulary_map; }
104 inline static const vector<string> reserved_tokens = {PAD_TOKEN, UNK_TOKEN, START_TOKEN, END_TOKEN};
Dataset()=default
virtual Index get_samples_number() const
Returns the total number of samples (rows) in the data matrix.
Definition dataset.h:74
Thread-safe positional file reader (pread on POSIX, overlapped ReadFile on Windows).
Definition io_utilities.h:20
Definition json.h:72
Definition json.h:85
VectorI calculate_target_distribution() const override
Returns the distribution of target tokens or classes.
const unordered_map< Index, string > & get_target_inverse_vocabulary_map() const
Definition language_dataset.h:40
void fill_inputs(const vector< Index > &, const vector< Index > &, float *, bool is_training, bool parallelize=true, int=-1) const override
Streams input token indices of the selected samples into the destination buffer.
const vector< string > & get_input_vocabulary() const
Definition language_dataset.h:32
const vector< string > & get_target_vocabulary() const
Definition language_dataset.h:33
LanguageDataset(const filesystem::path &="")
Creates a language dataset, optionally reading documents from the given path.
void set_input_vocabulary(const vector< string > &)
Replaces the input vocabulary and rebuilds its lookup map.
void fill_decoder(const vector< Index > &, const vector< Index > &, float *, bool is_training, bool parallelize=true, int=-1) const override
Streams decoder token indices (target shifted right) into the destination buffer.
void to_JSON(JsonWriter &) const override
Writes dataset state to a JSON writer.
void from_JSON(const JsonDocument &) override
Loads dataset state from a JSON document.
void set_target_vocabulary(const vector< string > &)
Replaces the target vocabulary and rebuilds its lookup maps.
const unordered_map< string, Index > & get_target_vocabulary_map() const
Definition language_dataset.h:39
void read_txt()
Reads the configured text corpus into the dataset and builds vocabularies.
void fill_targets(const vector< Index > &, const vector< Index > &, float *, bool is_training, bool parallelize=true, int=-1) const override
Streams target token indices of the selected samples into the destination buffer.
static const vector< string > reserved_tokens
Definition language_dataset.h:104
LanguageDataset(const Index, Index, Index)
Creates a synthetic language dataset with the given sample count and vocabulary controls.
void set_maximum_vocabulary_size(Index new_maximum)
Definition language_dataset.h:50
void create_vocabulary(const vector< vector< string_view > > &, vector< string > &) const
Builds a vocabulary from the tokenized documents, filtered by frequency and size limits.
Index get_target_vocabulary_size() const
Definition language_dataset.h:36
LanguageDataset(const filesystem::path &, Index maximum_vocabulary_size, Index minimum_token_frequency)
Creates a language dataset with vocabulary size and minimum token frequency filters.
void set_minimum_token_frequency(Index new_minimum)
Definition language_dataset.h:51
Index get_maximum_input_sequence_length() const
Definition language_dataset.h:42
LanguageDataset(const filesystem::path &, Index maximum_vocabulary_size)
Creates a language dataset capped at the given vocabulary size.
Index get_input_vocabulary_size() const
Definition language_dataset.h:35
Index get_samples_number() const override
Returns the number of token sequences in the dataset.
Index get_maximum_target_sequence_length() const
Definition language_dataset.h:43
const unordered_map< string, Index > & get_input_vocabulary_map() const
Definition language_dataset.h:38
Definition adaptive_moment_estimation.h:14