Loading...
Searching...
No Matches
operators.h
Go to the documentation of this file.
87 vector<TensorView>& get_inputs(ForwardPropagation& fp, size_t layer, size_t i = 0) const noexcept
112 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
114 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
136 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
138 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
201 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
203 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
247 void set(Index new_input_features, Index new_output_features, Type new_weight_type = Type::FP32);
257 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
259 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
262 void apply(const TensorView& input, TensorView& output, cublasLtEpilogue_t epilogue = CUBLASLT_EPILOGUE_BIAS);
294 void link_parameters(span<const TensorView> views) override { combination.link_parameters(views); }
295 void link_gradients (span<const TensorView> views) override { combination.link_gradients(views); }
301 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
303 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
344 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
346 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
468 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
470 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
476 void apply_gpu(const TensorView& input, TensorView& output, cudnnActivationDescriptor_t fused_activation = nullptr);
509 void link_parameters(span<const TensorView> views) override { convolution.link_parameters(views); }
510 void link_gradients (span<const TensorView> views) override { convolution.link_gradients(views); }
523 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
525 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
554 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
556 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
610 void link_parameters(span<const TensorView> views) override { combination.link_parameters(views); }
611 void link_gradients (span<const TensorView> views) override { combination.link_gradients(views); }
617 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
619 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
675 // attention_output_slots = {ConcatenatedAttentionOutputs} (backward-only: merged output for SDPA)
683 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
685 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
846 void set(Index heads_number, Index query_sequence_length, Index head_dimension, Type compute_dtype);
849 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
855 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
911 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
913 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
916 void apply_cpu(const TensorView& input, TensorView& output, TensorView& maximal_indices, bool is_training);
940 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
942 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
981 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
983 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
997 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
999 void back_propagate(ForwardPropagation& fp, BackPropagation& bp, size_t layer) const noexcept override;
1014 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
1030 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
1046 void forward_propagate(ForwardPropagation& fp, size_t layer, bool is_training) noexcept override;
Definition json.h:85
Definition json.h:23
Definition adaptive_moment_estimation.h:14
VectorI maximal_indices(const VectorR &, Index)
Indices of the n largest elements of a vector.
Type
Numeric precision used for training or inference tensors.
Definition configuration.h:20
@ CUDNN_CONVOLUTION_FWD_ALGO_IMPLICIT_GEMM
Definition pch.h:103
Element-wise non-linear activation (Identity, Sigmoid, Tanh, ReLU, Softmax).
Definition operators.h:169
void apply_delta(const TensorView &outputs, TensorView &delta) const
Multiplies delta by the derivative of the activation evaluated at outputs.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void to_JSON(JsonWriter &w) const override
Serializes the operator configuration to a JSON writer.
static const string & to_string(Function function)
Returns the string name of a Function.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
vector< size_t > output_slots_backward
Definition operators.h:192
void apply_gpu(TensorView &output)
GPU forward implementation; applies the activation in place on output.
void set_function(Function new_function)
Selects the activation function and configures cuDNN descriptors.
ActivationOp(const ActivationOp &)=delete
static const EnumMap< Function > & map()
Returns the bidirectional mapping between Function values and their string names.
void set_function(const string &name)
Selects the activation function by name (delegates to set_function(Function)).
ActivationOp & operator=(const ActivationOp &)=delete
static cudnnActivationMode_t to_cudnn_mode(Function function)
Returns the cuDNN activation mode corresponding to a Function.
static Function from_string(const string &name)
Returns the Function corresponding to a string name.
void apply_cpu(TensorView &output)
CPU forward implementation; applies the activation in place on output.
void from_JSON(const Json *parent) override
Restores the operator configuration from a JSON node.
ActivationOp()=default
Element-wise sum of several input tensors (used by residual connections).
Definition operators.h:110
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
~AttentionOp() override
vector< TensorSpec > forward_scratch_specs(Index batch_size) const
Returns the tensor specs of the forward-pass scratch buffers used by the operator.
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
void set(Index heads_number, Index head_dimension, Index query_sequence_length, Index source_sequence_length, bool use_causal_mask, Type compute_dtype)
Configures the attention geometry and compute precision.
void set_dropout_rate(float rate)
Sets the post-softmax dropout rate (0 disables dropout).
Definition operators.h:666
void apply(const TensorView &query, const TensorView &key, const TensorView &value, const TensorView &source_input, TensorView &attention_weights, TensorView &attention_weights_dropped, TensorView &output, float *mask_scratch, bool is_training)
Computes attention output from Q, K, V; applies softmax, mask and dropout in place.
void apply_delta(const TensorView &query, const TensorView &key, const TensorView &value, const TensorView &attention_output, const TensorView &attention_weights, const TensorView &attention_weights_dropped, const TensorView &output_delta, TensorView &attention_weight_delta, TensorView &query_delta, TensorView &key_delta, TensorView &value_delta) const
Computes Q/K/V gradients from the output gradient and cached forward activations.
vector< size_t > attention_output_slots
Definition operators.h:680
void from_JSON(const Json *parent) override
Restores the operator configuration from a JSON node.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
AttentionOp()
AttentionOp(AttentionOp &&) noexcept
void to_JSON(JsonWriter &w) const override
Serializes the operator configuration to a JSON writer.
Workspace holding parameter gradients and per-layer deltas during a backward pass.
Definition back_propagation.h:21
vector< vector< TensorView > > delta_views
Definition back_propagation.h:44
Batch normalization with learnable scale/shift and running statistics for inference.
Definition operators.h:308
void load_state_from_JSON(const Json *parent) override
Restores persistent state (e.g. running statistics) from a JSON node.
void set(Index new_features, float new_momentum=0.1f)
Configures the per-feature normalization.
void link_states(span< const TensorView > views) override
Binds state views provided by the hosting layer.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
void init_defaults()
Resets gamma to one, beta to zero, and running stats to identity values.
void apply_delta(const TensorView &input, const TensorView &mean, const TensorView &inverse_variance, TensorView &delta) const
Computes the input gradient given the cached normalization statistics from forward.
void invalidate_inference_cache()
Marks the inference cache as stale so it is rebuilt on the next inference call.
Definition operators.h:362
vector< TensorSpec > state_specs() const override
Returns the tensor specs of persistent state owned by this operator.
void from_JSON(const Json *parent) override
Restores the operator configuration from a JSON node.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void update_inference_cache()
Rebuilds the fused scale/shift cache used by the inference path from running stats.
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:335
void set_parameters_random() override
Initializes parameters with random values.
Definition operators.h:334
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
bool active() const
Returns true when the operator has been configured (features > 0).
Definition operators.h:321
void to_JSON(JsonWriter &w) const override
Serializes the operator configuration to a JSON writer.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
Clamps each output channel to a configurable lower/upper interval.
Definition operators.h:1004
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
Method
Disables bounding or enables per-channel clamping.
Definition operators.h:1006
Owning raw byte buffer that lives on CPU or CUDA memory, with aligned (re)allocation.
Definition tensor_utilities.h:166
Affine combination output = input * weights + bias (the dense matmul building block).
Definition operators.h:232
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
void set_parameters_random() override
Initializes parameters with random values.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void apply(const TensorView &input, TensorView &output, cublasLtEpilogue_t epilogue=CUBLASLT_EPILOGUE_BIAS)
Computes output = input * weights + bias with an optional fused epilogue (ReLU, bias,...
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
void apply_delta(const TensorView &output_delta, const TensorView &input, TensorView &input_delta, bool accumulate_input_delta=false) const
Computes input_delta from output_delta and updates weight/bias gradients.
void set(Index new_input_features, Index new_output_features, Type new_weight_type=Type::FP32)
Configures input/output dimensions and the weight storage dtype.
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
Fused affine + ReLU activation (uses cuBLASLt epilogue on GPU when available).
Definition operators.h:286
void set(Index input_features, Index output_features, Type weight_type=Type::FP32)
Configures the underlying CombinationOp; ReLU is fixed.
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:298
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
Definition operators.h:294
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
Definition operators.h:293
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void set_parameters_random() override
Initializes parameters with random values.
Definition operators.h:297
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
Definition operators.h:295
2D convolution operator (NHWC layout) backed by Eigen on CPU and cuDNN on GPU.
Definition operators.h:397
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
void set(Index input_h, Index input_w, Index kernels_n, Index kernel_h, Index kernel_w, Index kernel_c, Index row_stride, Index column_stride, Index padding_h, Index padding_w, Type compute_dtype)
Configures the convolution geometry and compute precision.
ConvolutionOp & operator=(const ConvolutionOp &)=delete
ConvolutionOp(const ConvolutionOp &)=delete
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
void set_parameters_random() override
Initializes parameters with random values.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void apply_cpu(const TensorView &input, TensorView &output)
CPU forward path; im2col + GEMM.
ConvolutionOp()=default
void apply_delta(const TensorView &input, const TensorView &output_delta, TensorView &input_delta) const
Computes input_delta from output_delta and updates weight/bias gradients.
void apply_gpu(const TensorView &input, TensorView &output, cudnnActivationDescriptor_t fused_activation=nullptr)
GPU forward path; runs cuDNN convolution with an optional fused activation.
ConvolutionReluOp()=default
void set(Index input_h, Index input_w, Index kernels_n, Index kernel_h, Index kernel_w, Index kernel_c, Index row_stride, Index column_stride, Index padding_h, Index padding_w, Type compute_dtype)
Configures the underlying ConvolutionOp; ReLU is fixed.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
Definition operators.h:508
void set_parameters_random() override
Initializes parameters with random values.
Definition operators.h:512
~ConvolutionReluOp() override
Definition operators.h:516
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
Definition operators.h:510
ConvolutionReluOp(const ConvolutionReluOp &)=delete
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:513
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
Definition operators.h:515
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
Definition operators.h:509
ConvolutionReluOp & operator=(const ConvolutionReluOp &)=delete
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
Inverted dropout: at training time zeros activations with probability rate and rescales survivors.
Definition operators.h:122
bool active() const
Returns true when the dropout rate is non-zero.
Definition operators.h:130
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
DropoutOp(DropoutOp &&) noexcept=default
void apply_gpu(TensorView &output)
GPU forward implementation; samples the mask and rescales survivors in place.
void to_JSON(JsonWriter &w) const override
Serializes the operator configuration to a JSON writer.
void set_rate(float new_rate)
Sets the drop probability (0 disables dropout).
void from_JSON(const Json *parent) override
Restores the operator configuration from a JSON node.
void apply_cpu(TensorView &output)
CPU forward implementation; samples the mask and rescales survivors in place.
void apply_delta(TensorView &delta) const
Applies the cached mask to a gradient tensor during the backward pass.
DropoutOp()=default
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
Token embedding lookup with optional scaling and additive positional encoding.
Definition operators.h:947
bool add_positional_encoding
Definition operators.h:953
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
void init_positional_encoding()
Fills the positional-encoding state tensor with the standard sinusoidal pattern.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
TensorView positional_encoding
Definition operators.h:958
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
vector< TensorSpec > state_specs() const override
Returns the tensor specs of persistent state owned by this operator.
void link_states(span< const TensorView > views) override
Binds state views provided by the hosting layer.
void set(Index new_vocabulary_size, Index new_sequence_length, Index new_embedding_dimension)
Configures the lookup table dimensions.
void set_parameters_random() override
Initializes parameters with random values.
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition enum_map.h:18
Flattens a multi-dimensional tensor into a 2D (batch, features) tensor.
Definition operators.h:995
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
Workspace holding the activations of every layer during a forward pass.
Definition forward_propagation.h:20
vector< vector< vector< TensorView > > > views
Definition forward_propagation.h:45
Layer normalization with learnable scale/shift, applied across the embedding dimension.
Definition operators.h:530
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
void set(Index sequence_length, Index embedding_dimension)
Configures the operator for a (sequence_length, embedding_dimension) input.
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:548
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
void set_parameters_random() override
Initializes parameters with random values.
Definition operators.h:547
Reshapes (batch, heads, seq, head_dim) tensors back into (batch, seq, embed); no parameters.
Definition operators.h:839
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void set(Index heads_number, Index query_sequence_length, Index head_dimension, Type compute_dtype)
Configures the merge geometry.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
Projects (input_features) into (heads * head_dim) and reshapes for multi-head attention.
Definition operators.h:579
bool accumulate_input_delta_cross
Definition operators.h:600
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
vector< size_t > input_delta_slots_cross
Definition operators.h:598
vector< size_t > scratch_slots
Definition operators.h:592
vector< size_t > input_delta_slots_self
Definition operators.h:597
bool accumulate_input_delta_self
Definition operators.h:599
void apply_delta(const TensorView &head_delta, const TensorView &input, TensorView &input_delta, bool accumulate, float *scratch) const
Computes input_delta from per-head gradients and updates the projection weight gradient.
void set_parameters_random() override
Initializes parameters with random values.
Definition operators.h:613
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void apply(const TensorView &input, TensorView &head_output, float *scratch)
Projects input and reshapes the result into per-head form in head_output.
void link_parameters(span< const TensorView > views) override
Binds parameter views provided by the hosting layer.
Definition operators.h:610
void set_parameters_glorot() override
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:614
vector< TensorSpec > parameter_specs() const override
Returns the tensor specs of trainable parameters owned by this operator.
Definition operators.h:609
void link_gradients(span< const TensorView > views) override
Binds gradient views provided by the hosting layer.
Definition operators.h:611
void set(Index input_features, Index heads_number, Index head_dimension, Type compute_dtype)
Configures the projection geometry.
Base class for compute building blocks composed by layers (matmul, activation, dropout,...
Definition operators.h:28
virtual void destroy_cuda()
Releases CUDA resources owned by the operator; called from destructors.
Definition operators.h:74
virtual vector< TensorSpec > parameter_specs() const
Returns the tensor specs of trainable parameters owned by this operator.
Definition operators.h:32
TensorView & get_input(ForwardPropagation &fp, size_t layer, size_t i=0) const noexcept
Definition operators.h:82
virtual void link_parameters(span< const TensorView >)
Binds parameter views provided by the hosting layer.
Definition operators.h:38
TensorView & get_output(ForwardPropagation &fp, size_t layer, size_t i=0) const noexcept
Definition operators.h:92
virtual void back_propagate(ForwardPropagation &, BackPropagation &, size_t) const noexcept
Runs the operator's backward computation, accumulating into gradient/delta buffers.
Definition operators.h:62
virtual void from_JSON(const Json *)
Restores the operator configuration from a JSON node.
Definition operators.h:68
virtual void to_JSON(JsonWriter &) const
Serializes the operator configuration to a JSON writer.
Definition operators.h:65
virtual void link_gradients(span< const TensorView >)
Binds gradient views provided by the hosting layer.
Definition operators.h:41
virtual void load_state_from_JSON(const Json *)
Restores persistent state (e.g. running statistics) from a JSON node.
Definition operators.h:71
TensorView & get_output_delta(BackPropagation &bp, size_t layer, size_t i=0) const noexcept
Definition operators.h:97
vector< TensorView > & get_inputs(ForwardPropagation &fp, size_t layer, size_t i=0) const noexcept
Definition operators.h:87
TensorView & get_input_delta(BackPropagation &bp, size_t layer, size_t i=0) const noexcept
Definition operators.h:102
virtual void link_states(span< const TensorView >)
Binds state views provided by the hosting layer.
Definition operators.h:44
virtual void forward_propagate(ForwardPropagation &, size_t, bool) noexcept
Runs the operator's forward computation.
Definition operators.h:56
virtual void set_parameters_glorot()
Initializes parameters using Glorot (Xavier) initialization.
Definition operators.h:50
virtual ~Operator()=default
virtual void set_parameters_random()
Initializes parameters with random values.
Definition operators.h:47
virtual vector< TensorSpec > state_specs() const
Returns the tensor specs of persistent state owned by this operator.
Definition operators.h:35
Sequence-wide 1D pooling over the embedding dimension (mean or max).
Definition operators.h:930
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
void set(Index input_h, Index input_w, Index input_c, Index pool_h, Index pool_w, Index row_stride, Index column_stride, Index padding_h, Index padding_w, Method method)
Configures the pooling geometry.
void back_propagate(ForwardPropagation &fp, BackPropagation &bp, size_t layer) const noexcept override
Runs the operator's backward computation, accumulating into gradient/delta buffers.
PoolOp(const PoolOp &)=delete
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
PoolOp & operator=(const PoolOp &)=delete
PoolOp()=default
void destroy_cuda() override
Releases CUDA resources owned by the operator; called from destructors.
Scales inputs to a target range using per-feature minimum/maximum or mean/std statistics.
Definition operators.h:1019
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.
Non-owning view over a tensor: pointer, shape, and data type with rich reshape helpers.
Definition tensor_utilities.h:293
Inverse of ScaleOp: maps normalized outputs back to the original feature range.
Definition operators.h:1035
void forward_propagate(ForwardPropagation &fp, size_t layer, bool is_training) noexcept override
Runs the operator's forward computation.