2026-08-01 11:27:28 -04:00
//+------------------------------------------------------------------+
//| NeuronOCLLSTM.mqh |
//| |
//| CNeuronLSTMOCL - the accelerated sequence LSTM (fused kernels, |
//| BPTT caches). |
//| |
//| Included from AI\Network.mqh AFTER every class declaration - |
//| bodies only, no declarations. Relocation is behaviour-neutral by |
//| construction: nothing here is reachable until Network.mqh ends. |
//+------------------------------------------------------------------+
# ifndef WARRIOR_AI_IMPL_NEURONOCLLSTM_MQH
# define WARRIOR_AI_IMPL_NEURONOCLLSTM_MQH
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
CNeuronLSTMOCL : : ~ CNeuronLSTMOCL ( void )
{
if ( CheckPointer ( WeightsLSTM ) ! = POINTER_INVALID )
delete WeightsLSTM ;
if ( CheckPointer ( FirstMomentumLSTM ) ! = POINTER_INVALID )
delete FirstMomentumLSTM ;
if ( CheckPointer ( SecondMomentumLSTM ) ! = POINTER_INVALID )
delete SecondMomentumLSTM ;
if ( CheckPointer ( DeltaWeightsLSTM ) ! = POINTER_INVALID )
delete DeltaWeightsLSTM ;
if ( CheckPointer ( WeightsGradient ) ! = POINTER_INVALID )
delete WeightsGradient ;
if ( CheckPointer ( Concatenated ) ! = POINTER_INVALID )
delete Concatenated ;
if ( CheckPointer ( ConcatenatedGradient ) ! = POINTER_INVALID )
delete ConcatenatedGradient ;
if ( CheckPointer ( Memory ) ! = POINTER_INVALID )
delete Memory ;
if ( CheckPointer ( HiddenCache ) ! = POINTER_INVALID )
delete HiddenCache ;
if ( CheckPointer ( CacheGates ) ! = POINTER_INVALID )
delete CacheGates ;
if ( CheckPointer ( CacheCell ) ! = POINTER_INVALID )
delete CacheCell ;
if ( CheckPointer ( CacheHidden ) ! = POINTER_INVALID )
delete CacheHidden ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL ::Init ( uint numOutputs , uint myIndex , COpenCLMy * open_cl , uint numNeurons , ENUM_OPTIMIZATION optimization_type )
{
if ( ! CNeuronBaseOCL ::Init ( numOutputs , myIndex , open_cl , numNeurons , optimization_type ) )
return false ;
uint H = numNeurons ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
{
Memory = new CBufferDouble ( ) ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
}
if ( ! Memory . BufferInit ( 2 * H , 0 ) | | ! Memory . BufferCreate ( OpenCL ) )
return false ;
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
{
Concatenated = new CBufferDouble ( ) ;
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
return false ;
}
if ( ! Concatenated . BufferInit ( 4 * H , 0 ) | | ! Concatenated . BufferCreate ( OpenCL ) )
return false ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
{
ConcatenatedGradient = new CBufferDouble ( ) ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
return false ;
}
if ( ! ConcatenatedGradient . BufferInit ( 4 * H , 0 ) | | ! ConcatenatedGradient . BufferCreate ( OpenCL ) )
return false ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
{
HiddenCache = new CBufferDouble ( ) ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
return false ;
}
if ( ! HiddenCache . BufferInit ( H , 0 ) | | ! HiddenCache . BufferCreate ( OpenCL ) )
return false ;
m_iInputs = -1 ;
return true ;
}
//+------------------------------------------------------------------+
//| DirectML/D3D12 tier equivalent of Init(COpenCLMy*) above. |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL ::Init ( uint numOutputs , uint myIndex , CDirectMLMy * direct_ml , uint numNeurons , ENUM_OPTIMIZATION optimization_type )
{
if ( ! CNeuronBaseOCL ::Init ( numOutputs , myIndex , direct_ml , numNeurons , optimization_type ) )
return false ;
uint H = numNeurons ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
{
Memory = new CBufferDouble ( ) ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
}
if ( ! Memory . BufferInit ( 2 * H , 0 ) | | ! Memory . BufferCreate ( DirectML ) )
return false ;
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
{
Concatenated = new CBufferDouble ( ) ;
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
return false ;
}
if ( ! Concatenated . BufferInit ( 4 * H , 0 ) | | ! Concatenated . BufferCreate ( DirectML ) )
return false ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
{
ConcatenatedGradient = new CBufferDouble ( ) ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
return false ;
}
if ( ! ConcatenatedGradient . BufferInit ( 4 * H , 0 ) | | ! ConcatenatedGradient . BufferCreate ( DirectML ) )
return false ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
{
HiddenCache = new CBufferDouble ( ) ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
return false ;
}
if ( ! HiddenCache . BufferInit ( H , 0 ) | | ! HiddenCache . BufferCreate ( DirectML ) )
return false ;
m_iInputs = -1 ;
return true ;
}
//+------------------------------------------------------------------+
//| Lazily sized on the first feedForward call, once the previous |
//| layer's neuron count (the input size) is known. |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : SetInputs ( int count )
{
m_iInputs = count ;
uint H = ( uint ) Neurons ( ) ;
//--- Resolve the sequence shape BEFORE sizing the weights: in sequence mode the gates see only ONE
//--- timestep, so the weight block is 4H(H + stepInputs + 1), not 4H(H + count + 1). That is the
//--- whole parameter saving of a recurrence - the same weights are reused at every step instead of
//--- one gigantic block reading all T*Iw inputs at once. At H1 defaults on LSTM: 4*16*(16+21+1) =
//--- 2432 weights against the old 4*16*(16+420+1) = 27968.
m_iSteps = -1 ;
if ( m_iStepInputs > 0 )
{
if ( count % m_iStepInputs ! = 0 )
{
printf ( " CNeuronLSTMOCL::SetInputs: input width %d is not a whole number of %d-wide timesteps - falling back to single-timestep mode " , count , m_iStepInputs ) ;
m_iStepInputs = -1 ;
}
else
m_iSteps = count / m_iStepInputs ;
}
int gateInputs = IsSequenceMode ( ) ? m_iStepInputs : count ;
int total = ( int ) ( 4 * H * ( H + gateInputs + 1 ) ) ;
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID )
{
WeightsLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID )
return false ;
}
if ( ! WeightsLSTM .Reserve ( total ) )
return false ;
// Fan-in-scaled (LeCun-uniform) init - see CNeuronBaseOCL::Init's OpenCL overload for the full
// rationale; fan-in here is hidden units + input width (each gate reads both).
double weighScale = 1.0 / MathSqrt ( ( double ) ( H + gateInputs ) + 1.0 ) ;
for ( int i = 0 ; i < total ; i + + )
{
double weigh = ( ( MathRand ( ) + 1 ) / 32768.0 - 0.5 ) * 2.0 * weighScale ;
if ( weigh = = 0 )
weigh = 0.001 ;
if ( ! WeightsLSTM . Add ( weigh ) )
return false ;
}
//--- POSITIVE FORGET-GATE BIAS. The single most important initialization detail in a recurrent net,
//--- and the one that decides whether this layer is a sequence model or an expensive 1-bar model.
//--- With every weight drawn around zero the forget gate starts at sigmoid(0) = 0.5, so the cell
//--- state is HALVED every timestep: c_t = f*c_{t-1} + i*g. Over m_iSteps bars the first bar
//--- survives into the output scaled by ~0.5^T, and the gradient reaches it scaled by the same
//--- factor (dc_prev = dc_total * f). At T=20 that is ~1e-6 - the recurrence exists on paper and
//--- carries nothing. Measured with DirectML\lstm_seq_flowcheck.cpp at the shipped H1 shapes
//--- (H=64, stepInputs=21, T=20), influence of bar 0 on the output relative to bar 19:
//--- bias 0.0 -> 3.0e-05 forward, 3.3e-05 backward (dead: a one-bar model)
//--- bias 1.0 -> 1.2e-02 forward, 1.4e-02 backward
//--- bias 2.0 -> 2.5e-01 forward, 2.7e-01 backward (a genuinely 20-bar-wide receptive field)
//--- This is not a tuning knob discovered by trial: Gers/Schmidhuber/Cummins (2000) introduced the
//--- forget gate with a positive bias, and Jozefowicz/Zaremba/Sutskever (ICML 2015) found "adding a
//--- bias of 1 to the forget gate" closes the LSTM-vs-GRU gap and recommend it as a default (it is
//--- why Keras ships unit_forget_bias=True).
//--- LOWERED 2.0 -> 1.0 on 2026-07-31. Picking 2.0 off the sweep above was a mistake of method: the
//--- sweep measures gradient REACH, and reach is not the objective - it trades directly against
//--- saturation, which the sweep does not measure at all. Over T steps the cell tends to
//--- c* = i*g/(1-sigmoid(b)). At b=2, sigmoid=0.88, so c* ~ 8.3*i*g, |c*| reaches ~4.2 and tanh(c*)
//--- pins at 0.9995 with derivative ~1e-3: h_T = o*tanh(c) becomes near-binary and is set by the gate
//--- biases rather than by the bars. At b=1, sigmoid=0.73, c* ~ 3.7*i*g, |c*| ~ 1.85, tanh ~ 0.95 with
//--- derivative ~0.1 - saturating but alive. That is the difference between a layer that summarises the
//--- window and one that emits a constant, and "outputs do not vary with the input" is precisely the
//--- 2026-07-30 sequence-LSTM symptom (flat IS error, Neutral:100%) that was misread as a gradient fault.
//--- Diagnose with "OOS raw out B:min..max" in the era line: a collapsed span is this, not a dead gradient.
//--- Symptom when this regresses: IS error flat to 2 decimals across many eras with OOS recall
//--- Neutral:100%, because a model that cannot see across bars can only predict the base rate.
//--- Layout: gates are [forget, input, output, candidate]; forget is gate 0, each hidden unit owns a
//--- row of (H + gateInputs + 1) with the bias last. Keep in step with the layout note in feedForward.
int rowWidth = ( int ) H + gateInputs + 1 ;
for ( uint hid = 0 ; hid < H ; hid + + )
if ( ! WeightsLSTM . Update ( ( int ) hid * rowWidth + rowWidth - 1 , LSTM_FORGET_BIAS_INIT ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! WeightsLSTM . BufferCreate ( OpenCL ) : ! WeightsLSTM . BufferCreate ( DirectML ) )
return false ;
//---
if ( CheckPointer ( FirstMomentumLSTM ) = = POINTER_INVALID )
{
FirstMomentumLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( FirstMomentumLSTM ) = = POINTER_INVALID )
return false ;
}
if ( ! FirstMomentumLSTM . BufferInit ( total , 0 ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! FirstMomentumLSTM . BufferCreate ( OpenCL ) : ! FirstMomentumLSTM . BufferCreate ( DirectML ) )
return false ;
//---
if ( CheckPointer ( SecondMomentumLSTM ) = = POINTER_INVALID )
{
SecondMomentumLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( SecondMomentumLSTM ) = = POINTER_INVALID )
return false ;
}
if ( ! SecondMomentumLSTM . BufferInit ( total , 0 ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! SecondMomentumLSTM . BufferCreate ( OpenCL ) : ! SecondMomentumLSTM . BufferCreate ( DirectML ) )
return false ;
//---
// Momentum accumulator - only meaningfully used when optimization==SGD
// (see updateInputWeights()), but allocated unconditionally regardless of
// the configured optimizer, matching CNeuronBaseOCL::Init's pattern for
// the dense-layer DeltaWeights/FirstMomentum/SecondMomentum trio above.
if ( CheckPointer ( DeltaWeightsLSTM ) = = POINTER_INVALID )
{
DeltaWeightsLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( DeltaWeightsLSTM ) = = POINTER_INVALID )
return false ;
}
if ( ! DeltaWeightsLSTM . BufferInit ( total , 0 ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! DeltaWeightsLSTM . BufferCreate ( OpenCL ) : ! DeltaWeightsLSTM . BufferCreate ( DirectML ) )
return false ;
//---
if ( CheckPointer ( WeightsGradient ) = = POINTER_INVALID )
{
WeightsGradient = new CBufferDouble ( ) ;
if ( CheckPointer ( WeightsGradient ) = = POINTER_INVALID )
return false ;
}
if ( ! WeightsGradient . BufferInit ( total , 0 ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! WeightsGradient . BufferCreate ( OpenCL ) : ! WeightsGradient . BufferCreate ( DirectML ) )
return false ;
//--- Belt-and-braces on the state buffers. SetInputs() is the LAZY sizing path - it runs on the first
//--- feedForward whenever m_iInputs is still unset, which includes the just-loaded-a-never-run-net case
//--- (see the note in Load()). Every buffer LSTMGates/LSTMState touch must exist by the time this
//--- returns, or the layer fails silently on every pass; do not assume Init() or Load() got here first.
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
{
Memory = new CBufferDouble ( ) ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
}
if ( Memory . GetIndex ( ) < 0 )
{
if ( ! Memory . BufferInit ( 2 * H , 0 ) )
return false ;
if ( CheckPointer ( OpenCL ) ! = POINTER_INVALID ? ! Memory . BufferCreate ( OpenCL ) : ! Memory . BufferCreate ( DirectML ) )
return false ;
}
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID | | Concatenated . GetIndex ( ) < 0 | |
CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID | | ConcatenatedGradient . GetIndex ( ) < 0 | |
CheckPointer ( HiddenCache ) = = POINTER_INVALID | | HiddenCache . GetIndex ( ) < 0 )
{
//--- loud on purpose: the failure this replaces was a silent `return false` out of feedForward,
//--- which looked identical to a healthy net that simply never fires.
printf ( " CNeuronLSTMOCL::SetInputs: scratch buffers missing (H=%d I=%d) - layer built by neither Init() nor Load() " , ( int ) H , count ) ;
return false ;
}
//--- Per-timestep caches. Only sequence mode needs them, and they are pure scratch (recomputed by every
//--- forward pass), so they are never persisted - Load re-creates them from the saved shape.
if ( IsSequenceMode ( ) & & ! AllocateSequenceCaches ( ) )
return false ;
//---
return true ;
}
//+------------------------------------------------------------------+
//| (Re)allocates the per-timestep BPTT caches for the current shape. |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : AllocateSequenceCaches ( void )
{
if ( ! IsSequenceMode ( ) )
return true ;
uint H = ( uint ) Neurons ( ) ;
if ( CheckPointer ( CacheGates ) = = POINTER_INVALID )
{
CacheGates = new CBufferDouble ( ) ;
if ( CheckPointer ( CacheGates ) = = POINTER_INVALID )
return false ;
}
CacheGates . BufferFree ( ) ;
if ( ! CacheGates . BufferInit ( ( int ) ( m_iSteps * 4 * H ) , 0 ) | | ! BackendBufferCreate ( CacheGates ) )
return false ;
if ( CheckPointer ( CacheCell ) = = POINTER_INVALID )
{
CacheCell = new CBufferDouble ( ) ;
if ( CheckPointer ( CacheCell ) = = POINTER_INVALID )
return false ;
}
CacheCell . BufferFree ( ) ;
if ( ! CacheCell . BufferInit ( ( int ) ( m_iSteps * H ) , 0 ) | | ! BackendBufferCreate ( CacheCell ) )
return false ;
if ( CheckPointer ( CacheHidden ) = = POINTER_INVALID )
{
CacheHidden = new CBufferDouble ( ) ;
if ( CheckPointer ( CacheHidden ) = = POINTER_INVALID )
return false ;
}
CacheHidden . BufferFree ( ) ;
if ( ! CacheHidden . BufferInit ( ( int ) ( m_iSteps * H ) , 0 ) | | ! BackendBufferCreate ( CacheHidden ) )
return false ;
return true ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : feedForward ( CNeuronBaseOCL * NeuronOCL )
{
if ( CheckPointer ( NeuronOCL ) = = POINTER_INVALID )
return false ;
if ( m_iInputs < = 0 )
{
if ( ! SetInputs ( NeuronOCL . Neurons ( ) ) )
return false ;
}
else
if ( m_iInputs ! = NeuronOCL . Neurons ( ) )
return false ;
int H = Neurons ( ) ;
int I = m_iInputs ;
//--- Sequence mode: one fused call unrolls all T timesteps, sharing the gate weights across them and
//--- caching per-step state for BPTT. h_{-1}/c_{-1} are zero, so nothing leaks between samples.
if ( IsSequenceMode ( ) )
{
if ( CheckPointer ( DirectML ) ! = POINTER_INVALID )
{
if ( ! DirectML . LSTMSeqForward ( WeightsLSTM . GetIndex ( ) , NeuronOCL . getOutputIndex ( ) , CacheGates . GetIndex ( ) ,
CacheCell . GetIndex ( ) , CacheHidden . GetIndex ( ) , getOutputIndex ( ) ,
H , m_iStepInputs , m_iSteps ) )
{
printf ( " Error of execution LSTM sequence feedForward (H=%d step=%d steps=%d) " , H , m_iStepInputs , m_iSteps ) ;
return false ;
}
return Output . BufferRead ( ) ;
}
if ( CheckPointer ( OpenCL ) = = POINTER_INVALID )
return feedForwardCPU ( NeuronOCL ) ;
//--- One launch per timestep: each is an implicit global barrier, which is the only ordering
//--- guarantee OpenCL gives across work-groups. See LSTM_SeqStepForward in Network.cl.
uint offSeq [ 1 ] = { 0 } ;
uint szSeq [ 1 ] = { ( uint ) H } ;
for ( int st = 0 ; st < m_iSteps ; st + + )
{
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_inputs , NeuronOCL . getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_cache_gates , CacheGates . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_cache_cell , CacheCell . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_cache_hidden , CacheHidden . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepForward , def_k_lsf_output , getOutputIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepForward , def_k_lsf_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepForward , def_k_lsf_step_inputs , m_iStepInputs ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepForward , def_k_lsf_steps , m_iSteps ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepForward , def_k_lsf_t , st ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_SeqStepForward , 1 , offSeq , szSeq ) )
{
printf ( " Error of execution kernel LSTM_SeqStepForward (t=%d): %d " , st , GetLastError ( ) ) ;
return false ;
}
}
return Output . BufferRead ( ) ;
}
if ( CheckPointer ( DirectML ) ! = POINTER_INVALID )
{
if ( ! DirectML . LSTMGates ( WeightsLSTM . GetIndex ( ) , getOutputIndex ( ) , NeuronOCL . getOutputIndex ( ) , Concatenated . GetIndex ( ) , H , I ) | |
! DirectML . LSTMState ( Concatenated . GetIndex ( ) , Memory . GetIndex ( ) , getOutputIndex ( ) , HiddenCache . GetIndex ( ) , getOutputIndex ( ) , H ) )
{
2026-08-02 01:09:18 -04:00
Print ( __FUNCTION__ + " : " + DirectML . BackendName ( ) + " LSTM feedForward failed, error " + IntegerToString ( DirectML . LastError ( ) ) ) ;
2026-08-01 11:27:28 -04:00
return false ;
}
return Output . BufferRead ( ) ;
}
if ( CheckPointer ( OpenCL ) = = POINTER_INVALID )
return false ;
uint offset2 [ 2 ] = { 0 , 0 } ;
uint size2 [ 2 ] = { ( uint ) H , 4 } ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_Gates , def_k_lstmg_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_Gates , def_k_lstmg_hidden_prev , getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_Gates , def_k_lstmg_inputs , NeuronOCL . getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_Gates , def_k_lstmg_concatenated , Concatenated . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_Gates , def_k_lstmg_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_Gates , def_k_lstmg_input_size , I ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_Gates , 2 , offset2 , size2 ) )
{
printf ( " Error of execution kernel LSTM_Gates: %d " , GetLastError ( ) ) ;
return false ;
}
uint offset1 [ 1 ] = { 0 } ;
uint size1 [ 1 ] = { ( uint ) H } ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_State , def_k_lstms_concatenated , Concatenated . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_State , def_k_lstms_memory , Memory . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_State , def_k_lstms_hidden_prev , getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_State , def_k_lstms_hidden_cache , HiddenCache . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_State , def_k_lstms_output , getOutputIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_State , def_k_lstms_hidden_size , H ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_State , 1 , offset1 , size1 ) )
{
printf ( " Error of execution kernel LSTM_State: %d " , GetLastError ( ) ) ;
return false ;
}
//--- Output (== hidden_prev for the next timestep) stays GPU-resident; see the note in
//--- CNeuronBaseOCL::feedForward().
return true ;
}
//+------------------------------------------------------------------+
//| Pure-MQL5 double-precision mirror of Network.cl's LSTM_Gates + |
//| LSTM_State kernels, host buffers only (CPU inference). Recurrent |
//| state is carried EXACTLY as the kernels do: hidden_prev is this |
//| neuron's Output, the cell state is Memory[0..H); both persist |
//| across bars. Weight layout: 4 gates (forget,input,output,cand), |
//| each H*(H+I+1), per unit [H recurrent | I input | 1 bias]. |
//| NOTE: like every backend this advances state per call; over a |
//| long backtest fp rounding vs the training backend can drift, but |
//| within the validated per-step tolerance (see ValidateCpuInference)|
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : feedForwardCPU ( CNeuronBaseOCL * NeuronOCL )
{
if ( CheckPointer ( NeuronOCL ) = = POINTER_INVALID | | CheckPointer ( Output ) = = POINTER_INVALID | |
CheckPointer ( WeightsLSTM ) = = POINTER_INVALID | | CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
int H = Neurons ( ) ;
int I = m_iInputs ;
if ( I < = 0 | | NeuronOCL . Neurons ( ) ! = I | | H < = 0 )
return false ;
//--- Sequence mode mirror of CPU_LSTMSeqForward (WarriorCPU.cpp) - same gate order, same weight layout,
//--- same zero initial state. Inference only, so no caches are kept: only the final hidden state is
//--- needed. Keep this in step with the kernel; a divergence here shows up as a deployed model that
//--- trades differently from the one that was validated.
if ( IsSequenceMode ( ) )
{
int Iw = m_iStepInputs , steps = m_iSteps ;
int per_gate_s = H * ( H + Iw + 1 ) ;
if ( WeightsLSTM . Total ( ) < 4 * per_gate_s | | Output . Total ( ) < H )
return false ;
double hPrev [ ] , cPrev [ ] , gates_s [ ] ;
if ( ArrayResize ( hPrev , H ) ! = H | | ArrayResize ( cPrev , H ) ! = H | | ArrayResize ( gates_s , 4 ) ! = 4 )
return false ;
ArrayInitialize ( hPrev , 0.0 ) ;
ArrayInitialize ( cPrev , 0.0 ) ;
for ( int st = 0 ; st < steps ; st + + )
{
double hNext [ ] ;
if ( ArrayResize ( hNext , H ) ! = H )
return false ;
for ( int id = 0 ; id < H ; id + + )
{
for ( int gate = 0 ; gate < 4 ; gate + + )
{
int shift = gate * per_gate_s + id * ( H + Iw + 1 ) ;
double sum = 0.0 ;
for ( int k = 0 ; k < H ; k + + )
sum + = hPrev [ k ] * WeightsLSTM . At ( shift + k ) ;
for ( int k = 0 ; k < Iw ; k + + )
sum + = NeuronOCL . OutputHost ( st * Iw + k ) * WeightsLSTM . At ( shift + H + k ) ;
sum + = WeightsLSTM . At ( shift + H + Iw ) ;
gates_s [ gate ] = ( gate < 3 ) ? ( 1.0 / ( 1.0 + exp ( - sum ) ) ) : tanh ( sum ) ;
}
double c_t = gates_s [ 0 ] * cPrev [ id ] + gates_s [ 1 ] * gates_s [ 3 ] ;
cPrev [ id ] = c_t ;
hNext [ id ] = gates_s [ 2 ] * tanh ( c_t ) ;
}
ArrayCopy ( hPrev , hNext ) ;
}
for ( int id = 0 ; id < H ; id + + )
if ( ! Output . Update ( id , hPrev [ id ] ) )
return false ;
return true ;
}
int per_gate = H * ( H + I + 1 ) ;
if ( WeightsLSTM . Total ( ) < 4 * per_gate | | Memory . Total ( ) < 2 * H | | Output . Total ( ) < H )
return false ;
//--- Gates: read hidden_prev (this Output) and inputs (prev Output) BEFORE Output is overwritten by
//--- the state step below - exactly the two-kernel ordering (Concatenated is a separate buffer there).
double gates [ ] ;
if ( ArrayResize ( gates , 4 * H ) ! = 4 * H )
return false ;
for ( int gate = 0 ; gate < 4 ; gate + + )
for ( int id = 0 ; id < H ; id + + )
{
int shift = gate * per_gate + id * ( H + I + 1 ) ;
double sum = 0.0 ;
for ( int k = 0 ; k < H ; k + + )
sum + = Output . At ( k ) * WeightsLSTM . At ( shift + k ) ; // hidden_prev
for ( int k = 0 ; k < I ; k + + )
sum + = NeuronOCL . OutputHost ( k ) * WeightsLSTM . At ( shift + H + k ) ; // inputs
sum + = WeightsLSTM . At ( shift + H + I ) ; // bias
gates [ gate * H + id ] = ( gate < 3 ) ? ( 1.0 / ( 1.0 + exp ( - sum ) ) ) : tanh ( sum ) ;
}
//--- State: c_t = f*c_prev + i*g ; h = o*tanh(c_t). memory[H+id] keeps c_prev (unused in inference).
for ( int id = 0 ; id < H ; id + + )
{
double f = gates [ id ] ;
double ii = gates [ H + id ] ;
double o = gates [ 2 * H + id ] ;
double g = gates [ 3 * H + id ] ;
double c_prev = Memory . At ( id ) ;
double c_t = f * c_prev + ii * g ;
if ( ! Memory . Update ( H + id , c_prev ) | | ! Memory . Update ( id , c_t ) )
return false ;
if ( ! Output . Update ( id , o * tanh ( c_t ) ) )
return false ;
}
return true ;
}
//+------------------------------------------------------------------+
//| Writes the gradient into NeuronOCL (the earlier/input-side layer) |
//| - same inverted-call convention as CNeuronConvOCL::calcInputGradients.|
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : calcInputGradients ( CNeuronBaseOCL * NeuronOCL )
{
if ( CheckPointer ( NeuronOCL ) = = POINTER_INVALID | | m_iInputs < = 0 )
return false ;
int H = Neurons ( ) ;
int I = m_iInputs ;
//--- Sequence mode: one fused call walks t = T-1 .. 0, carrying dh and dc back through every step and
//--- accumulating dW across all of them. This is the path that did not exist before - the per-step
//--- kernels below have no way to take dc from the following step, so the recurrent gradient was
//--- simply absent and the layer learned as if each sample were a single timestep.
if ( IsSequenceMode ( ) )
{
if ( CheckPointer ( DirectML ) ! = POINTER_INVALID )
{
if ( ! DirectML . LSTMSeqBackward ( WeightsLSTM . GetIndex ( ) , NeuronOCL . getOutputIndex ( ) , CacheGates . GetIndex ( ) ,
CacheCell . GetIndex ( ) , CacheHidden . GetIndex ( ) , getGradientIndex ( ) ,
WeightsGradient . GetIndex ( ) , NeuronOCL . getGradientIndex ( ) ,
H , m_iStepInputs , m_iSteps ) )
{
printf ( " Error of execution LSTM sequence calcInputGradients (H=%d step=%d steps=%d) " , H , m_iStepInputs , m_iSteps ) ;
return false ;
}
double tempSeq [ ] ;
return NeuronOCL . getGradient ( tempSeq ) > 0 ;
}
if ( CheckPointer ( OpenCL ) = = POINTER_INVALID )
return false ; // pure-MQL5 tier is inference-only and never backpropagates
//--- BPTT, host-driven so each launch is a global barrier. Scratch reuse: the single-timestep
//--- buffers are idle in sequence mode, so ConcatenatedGradient (4H) carries the gate gradients,
//--- HiddenCache (H) carries dh and Memory (2H, first half) carries dc - no extra allocations.
int seqTotal = 4 * H * ( H + m_iStepInputs + 1 ) ;
if ( ! WeightsGradient . BufferInit ( seqTotal , 0 ) | | ! WeightsGradient . BufferWrite ( ) )
return false ; // dW ACCUMULATES over the steps below, so it must start at zero
if ( ! Memory . BufferInit ( 2 * H , 0 ) | | ! Memory . BufferWrite ( ) )
return false ; // dc_T = 0
uint offG [ 1 ] = { 0 } ;
uint szH [ 1 ] = { ( uint ) H } ;
uint szW [ 1 ] = { ( uint ) seqTotal } ;
uint szIn [ 1 ] = { ( uint ) ( m_iStepInputs + H ) } ;
for ( int st = m_iSteps - 1 ; st > = 0 ; st - - )
{
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_out_gradient , getGradientIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_dh_buf , HiddenCache . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_dc_buf , Memory . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_cache_gates , CacheGates . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_cache_cell , CacheCell . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_gate_grad , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_steps , m_iSteps ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepGateGrad , def_k_lsgg_t , st ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_SeqStepGateGrad , 1 , offG , szH ) )
{
printf ( " Error of execution kernel LSTM_SeqStepGateGrad (t=%d): %d " , st , GetLastError ( ) ) ;
return false ;
}
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_gate_grad , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_cache_hidden , CacheHidden . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_inputs , NeuronOCL . getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_weights_gradient , WeightsGradient . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_step_inputs , m_iStepInputs ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepWeightGrad , def_k_lswg_t , st ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_SeqStepWeightGrad , 1 , offG , szW ) )
{
printf ( " Error of execution kernel LSTM_SeqStepWeightGrad (t=%d): %d " , st , GetLastError ( ) ) ;
return false ;
}
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_gate_grad , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_inputs_gradient , NeuronOCL . getGradientIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_dh_buf , HiddenCache . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_step_inputs , m_iStepInputs ) ;
OpenCL . SetArgument ( def_k_LSTM_SeqStepInputGrad , def_k_lsig_t , st ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_SeqStepInputGrad , 1 , offG , szIn ) )
{
printf ( " Error of execution kernel LSTM_SeqStepInputGrad (t=%d): %d " , st , GetLastError ( ) ) ;
return false ;
}
}
double tempSeqCl [ ] ;
return NeuronOCL . getGradient ( tempSeqCl ) > 0 ;
}
if ( CheckPointer ( DirectML ) ! = POINTER_INVALID )
{
if ( ! DirectML . LSTMGateGradient ( getGradientIndex ( ) , Memory . GetIndex ( ) , Concatenated . GetIndex ( ) , ConcatenatedGradient . GetIndex ( ) , H ) | |
! DirectML . LSTMWeightsGradient ( ConcatenatedGradient . GetIndex ( ) , HiddenCache . GetIndex ( ) , NeuronOCL . getOutputIndex ( ) , WeightsGradient . GetIndex ( ) , H , I ) | |
! DirectML . LSTMInputsGradient ( ConcatenatedGradient . GetIndex ( ) , WeightsLSTM . GetIndex ( ) , NeuronOCL . getGradientIndex ( ) , H , I ) )
{
2026-08-02 01:09:18 -04:00
Print ( __FUNCTION__ + " : " + DirectML . BackendName ( ) + " LSTM calcInputGradients failed, error " + IntegerToString ( DirectML . LastError ( ) ) ) ;
2026-08-01 11:27:28 -04:00
return false ;
}
double temp [ ] ;
return NeuronOCL . getGradient ( temp ) > 0 ;
}
if ( CheckPointer ( OpenCL ) = = POINTER_INVALID )
return false ;
uint offset1 [ 1 ] = { 0 } ;
uint size1 [ 1 ] = { ( uint ) H } ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_GateGradient , def_k_lstmgg_gradient , getGradientIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_GateGradient , def_k_lstmgg_memory , Memory . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_GateGradient , def_k_lstmgg_concatenated , Concatenated . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_GateGradient , def_k_lstmgg_concatenated_gradient , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_GateGradient , def_k_lstmgg_hidden_size , H ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_GateGradient , 1 , offset1 , size1 ) )
{
printf ( " Error of execution kernel LSTM_GateGradient: %d " , GetLastError ( ) ) ;
return false ;
}
uint sizeW [ 1 ] = { ( uint ) WeightsLSTM . Total ( ) } ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_WeightsGradient , def_k_lstmwg_concatenated_gradient , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_WeightsGradient , def_k_lstmwg_hidden_cache , HiddenCache . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_WeightsGradient , def_k_lstmwg_inputs , NeuronOCL . getOutputIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_WeightsGradient , def_k_lstmwg_weights_gradient , WeightsGradient . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_WeightsGradient , def_k_lstmwg_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_WeightsGradient , def_k_lstmwg_input_size , I ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_WeightsGradient , 1 , offset1 , sizeW ) )
{
printf ( " Error of execution kernel LSTM_WeightsGradient: %d " , GetLastError ( ) ) ;
return false ;
}
uint sizeI [ 1 ] = { ( uint ) I } ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_InputsGradient , def_k_lstmig_concatenated_gradient , ConcatenatedGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_InputsGradient , def_k_lstmig_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_InputsGradient , def_k_lstmig_inputs_gradient , NeuronOCL . getGradientIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_InputsGradient , def_k_lstmig_hidden_size , H ) ;
OpenCL . SetArgument ( def_k_LSTM_InputsGradient , def_k_lstmig_input_size , I ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_InputsGradient , 1 , offset1 , sizeI ) )
{
printf ( " Error of execution kernel LSTM_InputsGradient: %d " , GetLastError ( ) ) ;
return false ;
}
//--- NeuronOCL's Gradient stays GPU-resident; see the note in
//--- CNeuronConvOCL::calcInputGradients().
return true ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : updateInputWeights ( CNeuronBaseOCL * NeuronOCL )
{
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID )
return false ;
int total = WeightsLSTM . Total ( ) ;
if ( CheckPointer ( DirectML ) ! = POINTER_INVALID )
{
if ( optimization = = SGD )
{
if ( ! DirectML . LSTMUpdateWeightsMomentum ( WeightsLSTM . GetIndex ( ) , WeightsGradient . GetIndex ( ) , DeltaWeightsLSTM . GetIndex ( ) , eta , alpha , total ,
0 ) )
{
2026-08-02 01:09:18 -04:00
Print ( __FUNCTION__ + " : " + DirectML . BackendName ( ) + " LSTM_UpdateWeightsMomentum failed, error " + IntegerToString ( DirectML . LastError ( ) ) ) ;
2026-08-01 11:27:28 -04:00
return false ;
}
}
else
{
double lt = eta * sqrt ( 1 - pow ( b2 , t ) ) / ( 1 - pow ( b1 , t ) ) ;
if ( ! DirectML . LSTMUpdateWeightsAdam ( WeightsLSTM . GetIndex ( ) , WeightsGradient . GetIndex ( ) , FirstMomentumLSTM . GetIndex ( ) , SecondMomentumLSTM . GetIndex ( ) , lt , b1 , b2 , total ) )
{
2026-08-02 01:09:18 -04:00
Print ( __FUNCTION__ + " : " + DirectML . BackendName ( ) + " LSTM_UpdateWeightsAdam failed, error " + IntegerToString ( DirectML . LastError ( ) ) ) ;
2026-08-01 11:27:28 -04:00
return false ;
}
t + + ;
}
//--- WeightsLSTM stays DLL-resident; feedForward reads it via GetIndex() (same as the OpenCL
//--- branch below). Save()/BlendWeightsFrom() BufferRead() on demand.
return true ;
}
if ( CheckPointer ( OpenCL ) = = POINTER_INVALID )
return false ;
uint offset1 [ 1 ] = { 0 } ;
uint size1 [ 1 ] = { ( uint ) total } ;
if ( optimization = = SGD )
{
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_weights_gradient , WeightsGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_matrix_dw , DeltaWeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_learning_rates , ( float ) eta ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_momentum , ( float ) alpha ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsMomentum , def_k_lstmuwm_optimizer , 0 ) ;
ResetLastError ( ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_UpdateWeightsMomentum , 1 , offset1 , size1 ) )
{
printf ( " Error of execution kernel LSTM_UpdateWeightsMomentum: %d " , GetLastError ( ) ) ;
return false ;
}
}
else
{
double lt = eta * sqrt ( 1 - pow ( b2 , t ) ) / ( 1 - pow ( b1 , t ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_matrix_w , WeightsLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_weights_gradient , WeightsGradient . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_matrix_m , FirstMomentumLSTM . GetIndex ( ) ) ;
OpenCL . SetArgumentBuffer ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_matrix_v , SecondMomentumLSTM . GetIndex ( ) ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_l , ( float ) lt ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_b1 , ( float ) b1 ) ;
OpenCL . SetArgument ( def_k_LSTM_UpdateWeightsAdam , def_k_lstmuwa_b2 , ( float ) b2 ) ;
ResetLastError ( ) ;
if ( ! OpenCL . Execute ( def_k_LSTM_UpdateWeightsAdam , 1 , offset1 , size1 ) )
{
printf ( " Error of execution kernel LSTM_UpdateWeightsAdam: %d " , GetLastError ( ) ) ;
return false ;
}
t + + ;
}
//--- WeightsLSTM stays GPU-resident; see the note in CNeuronBaseOCL::updateInputWeights().
return true ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : Save ( const int file_handle )
{
if ( ! CNeuronBaseOCL : : Save ( file_handle ) )
return false ;
//--- Format tag FIRST. The pre-sequence format opened with m_iInputs here, and its weight block is a
//--- different SHAPE - 4H(H+totalInputs+1) against the sequence layer's 4H(H+stepInputs+1) - so a
//--- silent misread would not just be wrong, it would be wrong by a factor of ~13 in element count
//--- and corrupt everything after it in the file. A distinctive value no legitimate old m_iInputs
//--- could take lets Load() reject those files cleanly instead. See LSTM_SEQ_SAVE_TAG.
if ( FileWriteInteger ( file_handle , LSTM_SEQ_SAVE_TAG , INT_VALUE ) < INT_VALUE )
return false ;
if ( FileWriteInteger ( file_handle , m_iInputs , INT_VALUE ) < INT_VALUE )
return false ;
if ( FileWriteInteger ( file_handle , m_iStepInputs , INT_VALUE ) < INT_VALUE )
return false ;
if ( m_iInputs < = 0 )
return true ;
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID | | ! WeightsLSTM . BufferRead ( ) | | ! WeightsLSTM . Save ( file_handle ) )
return false ;
if ( CheckPointer ( FirstMomentumLSTM ) = = POINTER_INVALID | | ! FirstMomentumLSTM . BufferRead ( ) | | ! FirstMomentumLSTM . Save ( file_handle ) )
return false ;
if ( CheckPointer ( SecondMomentumLSTM ) = = POINTER_INVALID | | ! SecondMomentumLSTM . BufferRead ( ) | | ! SecondMomentumLSTM . Save ( file_handle ) )
return false ;
if ( CheckPointer ( DeltaWeightsLSTM ) = = POINTER_INVALID | | ! DeltaWeightsLSTM . BufferRead ( ) | | ! DeltaWeightsLSTM . Save ( file_handle ) )
return false ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID | | ! Memory . BufferRead ( ) | | ! Memory . Save ( file_handle ) )
return false ;
//---
return true ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNeuronLSTMOCL : : Load ( const int file_handle )
{
if ( ! CNeuronBaseOCL : : Load ( file_handle ) )
return false ;
int savedTag = FileReadInteger ( file_handle , INT_VALUE ) ;
if ( savedTag ! = LSTM_SEQ_SAVE_TAG )
{
//--- Pre-sequence .nnw. Its LSTM weight block is shaped for the whole flattened input as one
//--- timestep and cannot be reinterpreted; refuse so CNet::Load fails cleanly and the caller
//--- rebuilds a fresh topology, rather than reading a differently-shaped buffer and every
//--- subsequent layer at the wrong offset.
printf ( " CNeuronLSTMOCL::Load: this model predates the sequence-LSTM rewrite (found %d, expected %d) - it must be retrained. Delete its .nnw (and _shadow.nnw) to start clean. " , savedTag , LSTM_SEQ_SAVE_TAG ) ;
return false ;
}
m_iInputs = FileReadInteger ( file_handle , INT_VALUE ) ;
m_iStepInputs = FileReadInteger ( file_handle , INT_VALUE ) ;
m_iSteps = ( m_iStepInputs > 0 & & m_iInputs > 0 & & ( m_iInputs % m_iStepInputs ) = = 0 )
? m_iInputs / m_iStepInputs : -1 ;
uint H = ( uint ) Neurons ( ) ;
//--- scratch buffers (not persisted) were sized for the placeholder Init() unit
//--- count; resize them now that the real neuron count is known.
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
{
Concatenated = new CBufferDouble ( ) ;
if ( CheckPointer ( Concatenated ) = = POINTER_INVALID )
return false ;
}
Concatenated . BufferFree ( ) ;
if ( ! Concatenated . BufferInit ( 4 * H , 0 ) )
return false ;
if ( ! BackendBufferCreate ( Concatenated ) )
return false ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
{
ConcatenatedGradient = new CBufferDouble ( ) ;
if ( CheckPointer ( ConcatenatedGradient ) = = POINTER_INVALID )
return false ;
}
ConcatenatedGradient . BufferFree ( ) ;
if ( ! ConcatenatedGradient . BufferInit ( 4 * H , 0 ) )
return false ;
if ( ! BackendBufferCreate ( ConcatenatedGradient ) )
return false ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
{
HiddenCache = new CBufferDouble ( ) ;
if ( CheckPointer ( HiddenCache ) = = POINTER_INVALID )
return false ;
}
HiddenCache . BufferFree ( ) ;
if ( ! HiddenCache . BufferInit ( H , 0 ) )
return false ;
if ( ! BackendBufferCreate ( HiddenCache ) )
return false ;
//--- Memory (the c_prev cell state) is allocated HERE, before the m_iInputs<=0 early return below,
//--- and NOT only alongside the weight buffers further down. Both Init() overloads allocate it
//--- unconditionally; Load() used to allocate it only on the m_iInputs>0 path and SetInputs() never
//--- allocates it at all. A net saved before its first feedForward (Save writes m_iInputs=-1 and
//--- omits every LSTM buffer - e.g. weights-reset then detach, which is exactly what a "reset and
//--- restart" click produces) therefore came back from Load with Memory==NULL. The lazy SetInputs()
//--- on the next feedForward rebuilt the weights but not Memory, so LSTMState() got a dead buffer and
//--- every forward AND backward pass failed - the layer computed nothing, the head sat at a constant
//--- 1.0 for all three classes, and the model was pinned to Neutral forever while the journal filled
2026-08-02 01:09:18 -04:00
//--- with "... LSTM feedForward failed" (that line read "Error of execution DirectML LSTM feedForward"
//--- until 2026-08-02, when it started naming the backend that actually failed rather than always
//--- saying DirectML). If m_iInputs>0 the Memory.Load() further
2026-08-01 11:27:28 -04:00
//--- down simply overwrites what we allocate here.
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
{
Memory = new CBufferDouble ( ) ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
}
Memory . BufferFree ( ) ;
if ( ! Memory . BufferInit ( 2 * H , 0 ) )
return false ;
if ( ! BackendBufferCreate ( Memory ) )
return false ;
//---
if ( m_iInputs < = 0 )
return true ;
//--- In sequence mode the gates read ONE timestep, so the weight block is sized on the per-step width.
int total = ( int ) ( 4 * H * ( H + ( IsSequenceMode ( ) ? m_iStepInputs : m_iInputs ) + 1 ) ) ;
//---
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID )
{
WeightsLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( WeightsLSTM ) = = POINTER_INVALID )
return false ;
}
if ( WeightsLSTM . GetIndex ( ) > = 0 )
WeightsLSTM . BufferFree ( ) ;
if ( ! WeightsLSTM . Load ( file_handle ) )
return false ;
if ( ! BackendBufferCreate ( WeightsLSTM ) )
return false ;
//---
if ( CheckPointer ( FirstMomentumLSTM ) = = POINTER_INVALID )
{
FirstMomentumLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( FirstMomentumLSTM ) = = POINTER_INVALID )
return false ;
}
if ( FirstMomentumLSTM . GetIndex ( ) > = 0 )
FirstMomentumLSTM . BufferFree ( ) ;
if ( ! FirstMomentumLSTM . Load ( file_handle ) )
return false ;
if ( ! BackendBufferCreate ( FirstMomentumLSTM ) )
return false ;
//---
if ( CheckPointer ( SecondMomentumLSTM ) = = POINTER_INVALID )
{
SecondMomentumLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( SecondMomentumLSTM ) = = POINTER_INVALID )
return false ;
}
if ( SecondMomentumLSTM . GetIndex ( ) > = 0 )
SecondMomentumLSTM . BufferFree ( ) ;
if ( ! SecondMomentumLSTM . Load ( file_handle ) )
return false ;
if ( ! BackendBufferCreate ( SecondMomentumLSTM ) )
return false ;
//---
if ( CheckPointer ( DeltaWeightsLSTM ) = = POINTER_INVALID )
{
DeltaWeightsLSTM = new CBufferDouble ( ) ;
if ( CheckPointer ( DeltaWeightsLSTM ) = = POINTER_INVALID )
return false ;
}
if ( DeltaWeightsLSTM . GetIndex ( ) > = 0 )
DeltaWeightsLSTM . BufferFree ( ) ;
if ( ! DeltaWeightsLSTM . Load ( file_handle ) )
return false ;
if ( ! BackendBufferCreate ( DeltaWeightsLSTM ) )
return false ;
//---
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
{
Memory = new CBufferDouble ( ) ;
if ( CheckPointer ( Memory ) = = POINTER_INVALID )
return false ;
}
if ( Memory . GetIndex ( ) > = 0 )
Memory . BufferFree ( ) ;
if ( ! Memory . Load ( file_handle ) )
return false ;
if ( ! BackendBufferCreate ( Memory ) )
return false ;
//---
if ( CheckPointer ( WeightsGradient ) = = POINTER_INVALID )
{
WeightsGradient = new CBufferDouble ( ) ;
if ( CheckPointer ( WeightsGradient ) = = POINTER_INVALID )
return false ;
}
if ( ! WeightsGradient . BufferInit ( total , 0 ) )
return false ;
if ( ! BackendBufferCreate ( WeightsGradient ) )
return false ;
//--- Per-timestep BPTT caches are scratch and never persisted - rebuild them from the loaded shape.
if ( ! AllocateSequenceCaches ( ) )
return false ;
//---
return true ;
}
# endif