//+------------------------------------------------------------------+ //| NeuronOCLLSTM.mqh | //| | //| CNeuronLSTMOCL - the accelerated sequence LSTM (fused kernels, | //| BPTT caches). | //| | //| Included from AI\Network.mqh AFTER every class declaration - | //| bodies only, no declarations. Relocation is behaviour-neutral by | //| construction: nothing here is reachable until Network.mqh ends. | //+------------------------------------------------------------------+ #ifndef WARRIOR_AI_IMPL_NEURONOCLLSTM_MQH #define WARRIOR_AI_IMPL_NEURONOCLLSTM_MQH //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ CNeuronLSTMOCL::~CNeuronLSTMOCL(void) { if(CheckPointer(WeightsLSTM) != POINTER_INVALID) delete WeightsLSTM; if(CheckPointer(FirstMomentumLSTM) != POINTER_INVALID) delete FirstMomentumLSTM; if(CheckPointer(SecondMomentumLSTM) != POINTER_INVALID) delete SecondMomentumLSTM; if(CheckPointer(DeltaWeightsLSTM) != POINTER_INVALID) delete DeltaWeightsLSTM; if(CheckPointer(WeightsGradient) != POINTER_INVALID) delete WeightsGradient; if(CheckPointer(Concatenated) != POINTER_INVALID) delete Concatenated; if(CheckPointer(ConcatenatedGradient) != POINTER_INVALID) delete ConcatenatedGradient; if(CheckPointer(Memory) != POINTER_INVALID) delete Memory; if(CheckPointer(HiddenCache) != POINTER_INVALID) delete HiddenCache; if(CheckPointer(CacheGates) != POINTER_INVALID) delete CacheGates; if(CheckPointer(CacheCell) != POINTER_INVALID) delete CacheCell; if(CheckPointer(CacheHidden) != POINTER_INVALID) delete CacheHidden; } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::Init(uint numOutputs, uint myIndex, COpenCLMy *open_cl, uint numNeurons, ENUM_OPTIMIZATION optimization_type) { if(!CNeuronBaseOCL::Init(numOutputs, myIndex, open_cl, numNeurons, optimization_type)) return false; uint H = numNeurons; if(CheckPointer(Memory) == POINTER_INVALID) { Memory = new CBufferDouble(); if(CheckPointer(Memory) == POINTER_INVALID) return false; } if(!Memory.BufferInit(2 * H, 0) || !Memory.BufferCreate(OpenCL)) return false; if(CheckPointer(Concatenated) == POINTER_INVALID) { Concatenated = new CBufferDouble(); if(CheckPointer(Concatenated) == POINTER_INVALID) return false; } if(!Concatenated.BufferInit(4 * H, 0) || !Concatenated.BufferCreate(OpenCL)) return false; if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) { ConcatenatedGradient = new CBufferDouble(); if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) return false; } if(!ConcatenatedGradient.BufferInit(4 * H, 0) || !ConcatenatedGradient.BufferCreate(OpenCL)) return false; if(CheckPointer(HiddenCache) == POINTER_INVALID) { HiddenCache = new CBufferDouble(); if(CheckPointer(HiddenCache) == POINTER_INVALID) return false; } if(!HiddenCache.BufferInit(H, 0) || !HiddenCache.BufferCreate(OpenCL)) return false; m_iInputs = -1; return true; } //+------------------------------------------------------------------+ //| DirectML/D3D12 tier equivalent of Init(COpenCLMy*) above. | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::Init(uint numOutputs, uint myIndex, CDirectMLMy *direct_ml, uint numNeurons, ENUM_OPTIMIZATION optimization_type) { if(!CNeuronBaseOCL::Init(numOutputs, myIndex, direct_ml, numNeurons, optimization_type)) return false; uint H = numNeurons; if(CheckPointer(Memory) == POINTER_INVALID) { Memory = new CBufferDouble(); if(CheckPointer(Memory) == POINTER_INVALID) return false; } if(!Memory.BufferInit(2 * H, 0) || !Memory.BufferCreate(DirectML)) return false; if(CheckPointer(Concatenated) == POINTER_INVALID) { Concatenated = new CBufferDouble(); if(CheckPointer(Concatenated) == POINTER_INVALID) return false; } if(!Concatenated.BufferInit(4 * H, 0) || !Concatenated.BufferCreate(DirectML)) return false; if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) { ConcatenatedGradient = new CBufferDouble(); if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) return false; } if(!ConcatenatedGradient.BufferInit(4 * H, 0) || !ConcatenatedGradient.BufferCreate(DirectML)) return false; if(CheckPointer(HiddenCache) == POINTER_INVALID) { HiddenCache = new CBufferDouble(); if(CheckPointer(HiddenCache) == POINTER_INVALID) return false; } if(!HiddenCache.BufferInit(H, 0) || !HiddenCache.BufferCreate(DirectML)) return false; m_iInputs = -1; return true; } //+------------------------------------------------------------------+ //| Lazily sized on the first feedForward call, once the previous | //| layer's neuron count (the input size) is known. | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::SetInputs(int count) { m_iInputs = count; uint H = (uint)Neurons(); //--- Resolve the sequence shape BEFORE sizing the weights: in sequence mode the gates see only ONE //--- timestep, so the weight block is 4H(H + stepInputs + 1), not 4H(H + count + 1). That is the //--- whole parameter saving of a recurrence - the same weights are reused at every step instead of //--- one gigantic block reading all T*Iw inputs at once. At H1 defaults on LSTM: 4*16*(16+21+1) = //--- 2432 weights against the old 4*16*(16+420+1) = 27968. m_iSteps = -1; if(m_iStepInputs > 0) { if(count % m_iStepInputs != 0) { printf("CNeuronLSTMOCL::SetInputs: input width %d is not a whole number of %d-wide timesteps - falling back to single-timestep mode", count, m_iStepInputs); m_iStepInputs = -1; } else m_iSteps = count / m_iStepInputs; } int gateInputs = IsSequenceMode() ? m_iStepInputs : count; int total = (int)(4 * H * (H + gateInputs + 1)); if(CheckPointer(WeightsLSTM) == POINTER_INVALID) { WeightsLSTM = new CBufferDouble(); if(CheckPointer(WeightsLSTM) == POINTER_INVALID) return false; } if(!WeightsLSTM.Reserve(total)) return false; // Fan-in-scaled (LeCun-uniform) init - see CNeuronBaseOCL::Init's OpenCL overload for the full // rationale; fan-in here is hidden units + input width (each gate reads both). double weighScale = 1.0 / MathSqrt((double)(H + gateInputs) + 1.0); for(int i = 0; i < total; i++) { double weigh = ((MathRand() + 1) / 32768.0 - 0.5) * 2.0 * weighScale; if(weigh == 0) weigh = 0.001; if(!WeightsLSTM.Add(weigh)) return false; } //--- POSITIVE FORGET-GATE BIAS. The single most important initialization detail in a recurrent net, //--- and the one that decides whether this layer is a sequence model or an expensive 1-bar model. //--- With every weight drawn around zero the forget gate starts at sigmoid(0) = 0.5, so the cell //--- state is HALVED every timestep: c_t = f*c_{t-1} + i*g. Over m_iSteps bars the first bar //--- survives into the output scaled by ~0.5^T, and the gradient reaches it scaled by the same //--- factor (dc_prev = dc_total * f). At T=20 that is ~1e-6 - the recurrence exists on paper and //--- carries nothing. Measured with DirectML\lstm_seq_flowcheck.cpp at the shipped H1 shapes //--- (H=64, stepInputs=21, T=20), influence of bar 0 on the output relative to bar 19: //--- bias 0.0 -> 3.0e-05 forward, 3.3e-05 backward (dead: a one-bar model) //--- bias 1.0 -> 1.2e-02 forward, 1.4e-02 backward //--- bias 2.0 -> 2.5e-01 forward, 2.7e-01 backward (a genuinely 20-bar-wide receptive field) //--- This is not a tuning knob discovered by trial: Gers/Schmidhuber/Cummins (2000) introduced the //--- forget gate with a positive bias, and Jozefowicz/Zaremba/Sutskever (ICML 2015) found "adding a //--- bias of 1 to the forget gate" closes the LSTM-vs-GRU gap and recommend it as a default (it is //--- why Keras ships unit_forget_bias=True). //--- LOWERED 2.0 -> 1.0 on 2026-07-31. Picking 2.0 off the sweep above was a mistake of method: the //--- sweep measures gradient REACH, and reach is not the objective - it trades directly against //--- saturation, which the sweep does not measure at all. Over T steps the cell tends to //--- c* = i*g/(1-sigmoid(b)). At b=2, sigmoid=0.88, so c* ~ 8.3*i*g, |c*| reaches ~4.2 and tanh(c*) //--- pins at 0.9995 with derivative ~1e-3: h_T = o*tanh(c) becomes near-binary and is set by the gate //--- biases rather than by the bars. At b=1, sigmoid=0.73, c* ~ 3.7*i*g, |c*| ~ 1.85, tanh ~ 0.95 with //--- derivative ~0.1 - saturating but alive. That is the difference between a layer that summarises the //--- window and one that emits a constant, and "outputs do not vary with the input" is precisely the //--- 2026-07-30 sequence-LSTM symptom (flat IS error, Neutral:100%) that was misread as a gradient fault. //--- Diagnose with "OOS raw out B:min..max" in the era line: a collapsed span is this, not a dead gradient. //--- Symptom when this regresses: IS error flat to 2 decimals across many eras with OOS recall //--- Neutral:100%, because a model that cannot see across bars can only predict the base rate. //--- Layout: gates are [forget, input, output, candidate]; forget is gate 0, each hidden unit owns a //--- row of (H + gateInputs + 1) with the bias last. Keep in step with the layout note in feedForward. int rowWidth = (int)H + gateInputs + 1; for(uint hid = 0; hid < H; hid++) if(!WeightsLSTM.Update((int)hid * rowWidth + rowWidth - 1, LSTM_FORGET_BIAS_INIT)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !WeightsLSTM.BufferCreate(OpenCL) : !WeightsLSTM.BufferCreate(DirectML)) return false; //--- if(CheckPointer(FirstMomentumLSTM) == POINTER_INVALID) { FirstMomentumLSTM = new CBufferDouble(); if(CheckPointer(FirstMomentumLSTM) == POINTER_INVALID) return false; } if(!FirstMomentumLSTM.BufferInit(total, 0)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !FirstMomentumLSTM.BufferCreate(OpenCL) : !FirstMomentumLSTM.BufferCreate(DirectML)) return false; //--- if(CheckPointer(SecondMomentumLSTM) == POINTER_INVALID) { SecondMomentumLSTM = new CBufferDouble(); if(CheckPointer(SecondMomentumLSTM) == POINTER_INVALID) return false; } if(!SecondMomentumLSTM.BufferInit(total, 0)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !SecondMomentumLSTM.BufferCreate(OpenCL) : !SecondMomentumLSTM.BufferCreate(DirectML)) return false; //--- // Momentum accumulator - only meaningfully used when optimization==SGD // (see updateInputWeights()), but allocated unconditionally regardless of // the configured optimizer, matching CNeuronBaseOCL::Init's pattern for // the dense-layer DeltaWeights/FirstMomentum/SecondMomentum trio above. if(CheckPointer(DeltaWeightsLSTM) == POINTER_INVALID) { DeltaWeightsLSTM = new CBufferDouble(); if(CheckPointer(DeltaWeightsLSTM) == POINTER_INVALID) return false; } if(!DeltaWeightsLSTM.BufferInit(total, 0)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !DeltaWeightsLSTM.BufferCreate(OpenCL) : !DeltaWeightsLSTM.BufferCreate(DirectML)) return false; //--- if(CheckPointer(WeightsGradient) == POINTER_INVALID) { WeightsGradient = new CBufferDouble(); if(CheckPointer(WeightsGradient) == POINTER_INVALID) return false; } if(!WeightsGradient.BufferInit(total, 0)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !WeightsGradient.BufferCreate(OpenCL) : !WeightsGradient.BufferCreate(DirectML)) return false; //--- Belt-and-braces on the state buffers. SetInputs() is the LAZY sizing path - it runs on the first //--- feedForward whenever m_iInputs is still unset, which includes the just-loaded-a-never-run-net case //--- (see the note in Load()). Every buffer LSTMGates/LSTMState touch must exist by the time this //--- returns, or the layer fails silently on every pass; do not assume Init() or Load() got here first. if(CheckPointer(Memory) == POINTER_INVALID) { Memory = new CBufferDouble(); if(CheckPointer(Memory) == POINTER_INVALID) return false; } if(Memory.GetIndex() < 0) { if(!Memory.BufferInit(2 * H, 0)) return false; if(CheckPointer(OpenCL) != POINTER_INVALID ? !Memory.BufferCreate(OpenCL) : !Memory.BufferCreate(DirectML)) return false; } if(CheckPointer(Concatenated) == POINTER_INVALID || Concatenated.GetIndex() < 0 || CheckPointer(ConcatenatedGradient) == POINTER_INVALID || ConcatenatedGradient.GetIndex() < 0 || CheckPointer(HiddenCache) == POINTER_INVALID || HiddenCache.GetIndex() < 0) { //--- loud on purpose: the failure this replaces was a silent `return false` out of feedForward, //--- which looked identical to a healthy net that simply never fires. printf("CNeuronLSTMOCL::SetInputs: scratch buffers missing (H=%d I=%d) - layer built by neither Init() nor Load()", (int)H, count); return false; } //--- Per-timestep caches. Only sequence mode needs them, and they are pure scratch (recomputed by every //--- forward pass), so they are never persisted - Load re-creates them from the saved shape. if(IsSequenceMode() && !AllocateSequenceCaches()) return false; //--- return true; } //+------------------------------------------------------------------+ //| (Re)allocates the per-timestep BPTT caches for the current shape. | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::AllocateSequenceCaches(void) { if(!IsSequenceMode()) return true; uint H = (uint)Neurons(); if(CheckPointer(CacheGates) == POINTER_INVALID) { CacheGates = new CBufferDouble(); if(CheckPointer(CacheGates) == POINTER_INVALID) return false; } CacheGates.BufferFree(); if(!CacheGates.BufferInit((int)(m_iSteps * 4 * H), 0) || !BackendBufferCreate(CacheGates)) return false; if(CheckPointer(CacheCell) == POINTER_INVALID) { CacheCell = new CBufferDouble(); if(CheckPointer(CacheCell) == POINTER_INVALID) return false; } CacheCell.BufferFree(); if(!CacheCell.BufferInit((int)(m_iSteps * H), 0) || !BackendBufferCreate(CacheCell)) return false; if(CheckPointer(CacheHidden) == POINTER_INVALID) { CacheHidden = new CBufferDouble(); if(CheckPointer(CacheHidden) == POINTER_INVALID) return false; } CacheHidden.BufferFree(); if(!CacheHidden.BufferInit((int)(m_iSteps * H), 0) || !BackendBufferCreate(CacheHidden)) return false; return true; } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::feedForward(CNeuronBaseOCL *NeuronOCL) { if(CheckPointer(NeuronOCL) == POINTER_INVALID) return false; if(m_iInputs <= 0) { if(!SetInputs(NeuronOCL.Neurons())) return false; } else if(m_iInputs != NeuronOCL.Neurons()) return false; int H = Neurons(); int I = m_iInputs; //--- Sequence mode: one fused call unrolls all T timesteps, sharing the gate weights across them and //--- caching per-step state for BPTT. h_{-1}/c_{-1} are zero, so nothing leaks between samples. if(IsSequenceMode()) { if(CheckPointer(DirectML) != POINTER_INVALID) { if(!DirectML.LSTMSeqForward(WeightsLSTM.GetIndex(), NeuronOCL.getOutputIndex(), CacheGates.GetIndex(), CacheCell.GetIndex(), CacheHidden.GetIndex(), getOutputIndex(), H, m_iStepInputs, m_iSteps)) { printf("Error of execution LSTM sequence feedForward (H=%d step=%d steps=%d)", H, m_iStepInputs, m_iSteps); return false; } return Output.BufferRead(); } if(CheckPointer(OpenCL) == POINTER_INVALID) return feedForwardCPU(NeuronOCL); //--- One launch per timestep: each is an implicit global barrier, which is the only ordering //--- guarantee OpenCL gives across work-groups. See LSTM_SeqStepForward in Network.cl. uint offSeq[1] = {0}; uint szSeq[1] = {(uint)H}; for(int st = 0; st < m_iSteps; st++) { OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_inputs, NeuronOCL.getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_cache_gates, CacheGates.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_cache_cell, CacheCell.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_cache_hidden, CacheHidden.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepForward, def_k_lsf_output, getOutputIndex()); OpenCL.SetArgument(def_k_LSTM_SeqStepForward, def_k_lsf_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_SeqStepForward, def_k_lsf_step_inputs, m_iStepInputs); OpenCL.SetArgument(def_k_LSTM_SeqStepForward, def_k_lsf_steps, m_iSteps); OpenCL.SetArgument(def_k_LSTM_SeqStepForward, def_k_lsf_t, st); if(!OpenCL.Execute(def_k_LSTM_SeqStepForward, 1, offSeq, szSeq)) { printf("Error of execution kernel LSTM_SeqStepForward (t=%d): %d", st, GetLastError()); return false; } } return Output.BufferRead(); } if(CheckPointer(DirectML) != POINTER_INVALID) { if(!DirectML.LSTMGates(WeightsLSTM.GetIndex(), getOutputIndex(), NeuronOCL.getOutputIndex(), Concatenated.GetIndex(), H, I) || !DirectML.LSTMState(Concatenated.GetIndex(), Memory.GetIndex(), getOutputIndex(), HiddenCache.GetIndex(), getOutputIndex(), H)) { Print(__FUNCTION__ + ": " + DirectML.BackendName() + " LSTM feedForward failed, error " + IntegerToString(DirectML.LastError())); return false; } return Output.BufferRead(); } if(CheckPointer(OpenCL) == POINTER_INVALID) return false; uint offset2[2] = {0, 0}; uint size2[2] = {(uint)H, 4}; OpenCL.SetArgumentBuffer(def_k_LSTM_Gates, def_k_lstmg_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_Gates, def_k_lstmg_hidden_prev, getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_Gates, def_k_lstmg_inputs, NeuronOCL.getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_Gates, def_k_lstmg_concatenated, Concatenated.GetIndex()); OpenCL.SetArgument(def_k_LSTM_Gates, def_k_lstmg_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_Gates, def_k_lstmg_input_size, I); if(!OpenCL.Execute(def_k_LSTM_Gates, 2, offset2, size2)) { printf("Error of execution kernel LSTM_Gates: %d", GetLastError()); return false; } uint offset1[1] = {0}; uint size1[1] = {(uint)H}; OpenCL.SetArgumentBuffer(def_k_LSTM_State, def_k_lstms_concatenated, Concatenated.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_State, def_k_lstms_memory, Memory.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_State, def_k_lstms_hidden_prev, getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_State, def_k_lstms_hidden_cache, HiddenCache.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_State, def_k_lstms_output, getOutputIndex()); OpenCL.SetArgument(def_k_LSTM_State, def_k_lstms_hidden_size, H); if(!OpenCL.Execute(def_k_LSTM_State, 1, offset1, size1)) { printf("Error of execution kernel LSTM_State: %d", GetLastError()); return false; } //--- Output (== hidden_prev for the next timestep) stays GPU-resident; see the note in //--- CNeuronBaseOCL::feedForward(). return true; } //+------------------------------------------------------------------+ //| Pure-MQL5 double-precision mirror of Network.cl's LSTM_Gates + | //| LSTM_State kernels, host buffers only (CPU inference). Recurrent | //| state is carried EXACTLY as the kernels do: hidden_prev is this | //| neuron's Output, the cell state is Memory[0..H); both persist | //| across bars. Weight layout: 4 gates (forget,input,output,cand), | //| each H*(H+I+1), per unit [H recurrent | I input | 1 bias]. | //| NOTE: like every backend this advances state per call; over a | //| long backtest fp rounding vs the training backend can drift, but | //| within the validated per-step tolerance (see ValidateCpuInference)| //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::feedForwardCPU(CNeuronBaseOCL *NeuronOCL) { if(CheckPointer(NeuronOCL) == POINTER_INVALID || CheckPointer(Output) == POINTER_INVALID || CheckPointer(WeightsLSTM) == POINTER_INVALID || CheckPointer(Memory) == POINTER_INVALID) return false; int H = Neurons(); int I = m_iInputs; if(I <= 0 || NeuronOCL.Neurons() != I || H <= 0) return false; //--- Sequence mode mirror of CPU_LSTMSeqForward (WarriorCPU.cpp) - same gate order, same weight layout, //--- same zero initial state. Inference only, so no caches are kept: only the final hidden state is //--- needed. Keep this in step with the kernel; a divergence here shows up as a deployed model that //--- trades differently from the one that was validated. if(IsSequenceMode()) { int Iw = m_iStepInputs, steps = m_iSteps; int per_gate_s = H * (H + Iw + 1); if(WeightsLSTM.Total() < 4 * per_gate_s || Output.Total() < H) return false; double hPrev[], cPrev[], gates_s[]; if(ArrayResize(hPrev, H) != H || ArrayResize(cPrev, H) != H || ArrayResize(gates_s, 4) != 4) return false; ArrayInitialize(hPrev, 0.0); ArrayInitialize(cPrev, 0.0); for(int st = 0; st < steps; st++) { double hNext[]; if(ArrayResize(hNext, H) != H) return false; for(int id = 0; id < H; id++) { for(int gate = 0; gate < 4; gate++) { int shift = gate * per_gate_s + id * (H + Iw + 1); double sum = 0.0; for(int k = 0; k < H; k++) sum += hPrev[k] * WeightsLSTM.At(shift + k); for(int k = 0; k < Iw; k++) sum += NeuronOCL.OutputHost(st * Iw + k) * WeightsLSTM.At(shift + H + k); sum += WeightsLSTM.At(shift + H + Iw); gates_s[gate] = (gate < 3) ? (1.0 / (1.0 + exp(-sum))) : tanh(sum); } double c_t = gates_s[0] * cPrev[id] + gates_s[1] * gates_s[3]; cPrev[id] = c_t; hNext[id] = gates_s[2] * tanh(c_t); } ArrayCopy(hPrev, hNext); } for(int id = 0; id < H; id++) if(!Output.Update(id, hPrev[id])) return false; return true; } int per_gate = H * (H + I + 1); if(WeightsLSTM.Total() < 4 * per_gate || Memory.Total() < 2 * H || Output.Total() < H) return false; //--- Gates: read hidden_prev (this Output) and inputs (prev Output) BEFORE Output is overwritten by //--- the state step below - exactly the two-kernel ordering (Concatenated is a separate buffer there). double gates[]; if(ArrayResize(gates, 4 * H) != 4 * H) return false; for(int gate = 0; gate < 4; gate++) for(int id = 0; id < H; id++) { int shift = gate * per_gate + id * (H + I + 1); double sum = 0.0; for(int k = 0; k < H; k++) sum += Output.At(k) * WeightsLSTM.At(shift + k); // hidden_prev for(int k = 0; k < I; k++) sum += NeuronOCL.OutputHost(k) * WeightsLSTM.At(shift + H + k); // inputs sum += WeightsLSTM.At(shift + H + I); // bias gates[gate * H + id] = (gate < 3) ? (1.0 / (1.0 + exp(-sum))) : tanh(sum); } //--- State: c_t = f*c_prev + i*g ; h = o*tanh(c_t). memory[H+id] keeps c_prev (unused in inference). for(int id = 0; id < H; id++) { double f = gates[id]; double ii = gates[H + id]; double o = gates[2 * H + id]; double g = gates[3 * H + id]; double c_prev = Memory.At(id); double c_t = f * c_prev + ii * g; if(!Memory.Update(H + id, c_prev) || !Memory.Update(id, c_t)) return false; if(!Output.Update(id, o * tanh(c_t))) return false; } return true; } //+------------------------------------------------------------------+ //| Writes the gradient into NeuronOCL (the earlier/input-side layer) | //| - same inverted-call convention as CNeuronConvOCL::calcInputGradients.| //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::calcInputGradients(CNeuronBaseOCL *NeuronOCL) { if(CheckPointer(NeuronOCL) == POINTER_INVALID || m_iInputs <= 0) return false; int H = Neurons(); int I = m_iInputs; //--- Sequence mode: one fused call walks t = T-1 .. 0, carrying dh and dc back through every step and //--- accumulating dW across all of them. This is the path that did not exist before - the per-step //--- kernels below have no way to take dc from the following step, so the recurrent gradient was //--- simply absent and the layer learned as if each sample were a single timestep. if(IsSequenceMode()) { if(CheckPointer(DirectML) != POINTER_INVALID) { if(!DirectML.LSTMSeqBackward(WeightsLSTM.GetIndex(), NeuronOCL.getOutputIndex(), CacheGates.GetIndex(), CacheCell.GetIndex(), CacheHidden.GetIndex(), getGradientIndex(), WeightsGradient.GetIndex(), NeuronOCL.getGradientIndex(), H, m_iStepInputs, m_iSteps)) { printf("Error of execution LSTM sequence calcInputGradients (H=%d step=%d steps=%d)", H, m_iStepInputs, m_iSteps); return false; } double tempSeq[]; return NeuronOCL.getGradient(tempSeq) > 0; } if(CheckPointer(OpenCL) == POINTER_INVALID) return false; // pure-MQL5 tier is inference-only and never backpropagates //--- BPTT, host-driven so each launch is a global barrier. Scratch reuse: the single-timestep //--- buffers are idle in sequence mode, so ConcatenatedGradient (4H) carries the gate gradients, //--- HiddenCache (H) carries dh and Memory (2H, first half) carries dc - no extra allocations. int seqTotal = 4 * H * (H + m_iStepInputs + 1); if(!WeightsGradient.BufferInit(seqTotal, 0) || !WeightsGradient.BufferWrite()) return false; // dW ACCUMULATES over the steps below, so it must start at zero if(!Memory.BufferInit(2 * H, 0) || !Memory.BufferWrite()) return false; // dc_T = 0 uint offG[1] = {0}; uint szH[1] = {(uint)H}; uint szW[1] = {(uint)seqTotal}; uint szIn[1] = {(uint)(m_iStepInputs + H)}; for(int st = m_iSteps - 1; st >= 0; st--) { OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_out_gradient, getGradientIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_dh_buf, HiddenCache.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_dc_buf, Memory.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_cache_gates, CacheGates.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_cache_cell, CacheCell.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_gate_grad, ConcatenatedGradient.GetIndex()); OpenCL.SetArgument(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_steps, m_iSteps); OpenCL.SetArgument(def_k_LSTM_SeqStepGateGrad, def_k_lsgg_t, st); if(!OpenCL.Execute(def_k_LSTM_SeqStepGateGrad, 1, offG, szH)) { printf("Error of execution kernel LSTM_SeqStepGateGrad (t=%d): %d", st, GetLastError()); return false; } OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_gate_grad, ConcatenatedGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_cache_hidden, CacheHidden.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_inputs, NeuronOCL.getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_weights_gradient, WeightsGradient.GetIndex()); OpenCL.SetArgument(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_step_inputs, m_iStepInputs); OpenCL.SetArgument(def_k_LSTM_SeqStepWeightGrad, def_k_lswg_t, st); if(!OpenCL.Execute(def_k_LSTM_SeqStepWeightGrad, 1, offG, szW)) { printf("Error of execution kernel LSTM_SeqStepWeightGrad (t=%d): %d", st, GetLastError()); return false; } OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepInputGrad, def_k_lsig_gate_grad, ConcatenatedGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepInputGrad, def_k_lsig_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepInputGrad, def_k_lsig_inputs_gradient, NeuronOCL.getGradientIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_SeqStepInputGrad, def_k_lsig_dh_buf, HiddenCache.GetIndex()); OpenCL.SetArgument(def_k_LSTM_SeqStepInputGrad, def_k_lsig_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_SeqStepInputGrad, def_k_lsig_step_inputs, m_iStepInputs); OpenCL.SetArgument(def_k_LSTM_SeqStepInputGrad, def_k_lsig_t, st); if(!OpenCL.Execute(def_k_LSTM_SeqStepInputGrad, 1, offG, szIn)) { printf("Error of execution kernel LSTM_SeqStepInputGrad (t=%d): %d", st, GetLastError()); return false; } } double tempSeqCl[]; return NeuronOCL.getGradient(tempSeqCl) > 0; } if(CheckPointer(DirectML) != POINTER_INVALID) { if(!DirectML.LSTMGateGradient(getGradientIndex(), Memory.GetIndex(), Concatenated.GetIndex(), ConcatenatedGradient.GetIndex(), H) || !DirectML.LSTMWeightsGradient(ConcatenatedGradient.GetIndex(), HiddenCache.GetIndex(), NeuronOCL.getOutputIndex(), WeightsGradient.GetIndex(), H, I) || !DirectML.LSTMInputsGradient(ConcatenatedGradient.GetIndex(), WeightsLSTM.GetIndex(), NeuronOCL.getGradientIndex(), H, I)) { Print(__FUNCTION__ + ": " + DirectML.BackendName() + " LSTM calcInputGradients failed, error " + IntegerToString(DirectML.LastError())); return false; } double temp[]; return NeuronOCL.getGradient(temp) > 0; } if(CheckPointer(OpenCL) == POINTER_INVALID) return false; uint offset1[1] = {0}; uint size1[1] = {(uint)H}; OpenCL.SetArgumentBuffer(def_k_LSTM_GateGradient, def_k_lstmgg_gradient, getGradientIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_GateGradient, def_k_lstmgg_memory, Memory.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_GateGradient, def_k_lstmgg_concatenated, Concatenated.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_GateGradient, def_k_lstmgg_concatenated_gradient, ConcatenatedGradient.GetIndex()); OpenCL.SetArgument(def_k_LSTM_GateGradient, def_k_lstmgg_hidden_size, H); if(!OpenCL.Execute(def_k_LSTM_GateGradient, 1, offset1, size1)) { printf("Error of execution kernel LSTM_GateGradient: %d", GetLastError()); return false; } uint sizeW[1] = {(uint)WeightsLSTM.Total()}; OpenCL.SetArgumentBuffer(def_k_LSTM_WeightsGradient, def_k_lstmwg_concatenated_gradient, ConcatenatedGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_WeightsGradient, def_k_lstmwg_hidden_cache, HiddenCache.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_WeightsGradient, def_k_lstmwg_inputs, NeuronOCL.getOutputIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_WeightsGradient, def_k_lstmwg_weights_gradient, WeightsGradient.GetIndex()); OpenCL.SetArgument(def_k_LSTM_WeightsGradient, def_k_lstmwg_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_WeightsGradient, def_k_lstmwg_input_size, I); if(!OpenCL.Execute(def_k_LSTM_WeightsGradient, 1, offset1, sizeW)) { printf("Error of execution kernel LSTM_WeightsGradient: %d", GetLastError()); return false; } uint sizeI[1] = {(uint)I}; OpenCL.SetArgumentBuffer(def_k_LSTM_InputsGradient, def_k_lstmig_concatenated_gradient, ConcatenatedGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_InputsGradient, def_k_lstmig_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_InputsGradient, def_k_lstmig_inputs_gradient, NeuronOCL.getGradientIndex()); OpenCL.SetArgument(def_k_LSTM_InputsGradient, def_k_lstmig_hidden_size, H); OpenCL.SetArgument(def_k_LSTM_InputsGradient, def_k_lstmig_input_size, I); if(!OpenCL.Execute(def_k_LSTM_InputsGradient, 1, offset1, sizeI)) { printf("Error of execution kernel LSTM_InputsGradient: %d", GetLastError()); return false; } //--- NeuronOCL's Gradient stays GPU-resident; see the note in //--- CNeuronConvOCL::calcInputGradients(). return true; } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::updateInputWeights(CNeuronBaseOCL *NeuronOCL) { if(CheckPointer(WeightsLSTM) == POINTER_INVALID) return false; int total = WeightsLSTM.Total(); if(CheckPointer(DirectML) != POINTER_INVALID) { if(optimization == SGD) { if(!DirectML.LSTMUpdateWeightsMomentum(WeightsLSTM.GetIndex(), WeightsGradient.GetIndex(), DeltaWeightsLSTM.GetIndex(), eta, alpha, total, 0)) { Print(__FUNCTION__ + ": " + DirectML.BackendName() + " LSTM_UpdateWeightsMomentum failed, error " + IntegerToString(DirectML.LastError())); return false; } } else { double lt = eta * sqrt(1 - pow(b2, t)) / (1 - pow(b1, t)); if(!DirectML.LSTMUpdateWeightsAdam(WeightsLSTM.GetIndex(), WeightsGradient.GetIndex(), FirstMomentumLSTM.GetIndex(), SecondMomentumLSTM.GetIndex(), lt, b1, b2, total)) { Print(__FUNCTION__ + ": " + DirectML.BackendName() + " LSTM_UpdateWeightsAdam failed, error " + IntegerToString(DirectML.LastError())); return false; } t++; } //--- WeightsLSTM stays DLL-resident; feedForward reads it via GetIndex() (same as the OpenCL //--- branch below). Save()/BlendWeightsFrom() BufferRead() on demand. return true; } if(CheckPointer(OpenCL) == POINTER_INVALID) return false; uint offset1[1] = {0}; uint size1[1] = {(uint)total}; if(optimization == SGD) { OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_weights_gradient, WeightsGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_matrix_dw, DeltaWeightsLSTM.GetIndex()); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_learning_rates, (float)eta); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_momentum, (float)alpha); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsMomentum, def_k_lstmuwm_optimizer, 0); ResetLastError(); if(!OpenCL.Execute(def_k_LSTM_UpdateWeightsMomentum, 1, offset1, size1)) { printf("Error of execution kernel LSTM_UpdateWeightsMomentum: %d", GetLastError()); return false; } } else { double lt = eta * sqrt(1 - pow(b2, t)) / (1 - pow(b1, t)); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_matrix_w, WeightsLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_weights_gradient, WeightsGradient.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_matrix_m, FirstMomentumLSTM.GetIndex()); OpenCL.SetArgumentBuffer(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_matrix_v, SecondMomentumLSTM.GetIndex()); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_l, (float)lt); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_b1, (float)b1); OpenCL.SetArgument(def_k_LSTM_UpdateWeightsAdam, def_k_lstmuwa_b2, (float)b2); ResetLastError(); if(!OpenCL.Execute(def_k_LSTM_UpdateWeightsAdam, 1, offset1, size1)) { printf("Error of execution kernel LSTM_UpdateWeightsAdam: %d", GetLastError()); return false; } t++; } //--- WeightsLSTM stays GPU-resident; see the note in CNeuronBaseOCL::updateInputWeights(). return true; } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::Save(const int file_handle) { if(!CNeuronBaseOCL::Save(file_handle)) return false; //--- Format tag FIRST. The pre-sequence format opened with m_iInputs here, and its weight block is a //--- different SHAPE - 4H(H+totalInputs+1) against the sequence layer's 4H(H+stepInputs+1) - so a //--- silent misread would not just be wrong, it would be wrong by a factor of ~13 in element count //--- and corrupt everything after it in the file. A distinctive value no legitimate old m_iInputs //--- could take lets Load() reject those files cleanly instead. See LSTM_SEQ_SAVE_TAG. if(FileWriteInteger(file_handle, LSTM_SEQ_SAVE_TAG, INT_VALUE) < INT_VALUE) return false; if(FileWriteInteger(file_handle, m_iInputs, INT_VALUE) < INT_VALUE) return false; if(FileWriteInteger(file_handle, m_iStepInputs, INT_VALUE) < INT_VALUE) return false; if(m_iInputs <= 0) return true; if(CheckPointer(WeightsLSTM) == POINTER_INVALID || !WeightsLSTM.BufferRead() || !WeightsLSTM.Save(file_handle)) return false; if(CheckPointer(FirstMomentumLSTM) == POINTER_INVALID || !FirstMomentumLSTM.BufferRead() || !FirstMomentumLSTM.Save(file_handle)) return false; if(CheckPointer(SecondMomentumLSTM) == POINTER_INVALID || !SecondMomentumLSTM.BufferRead() || !SecondMomentumLSTM.Save(file_handle)) return false; if(CheckPointer(DeltaWeightsLSTM) == POINTER_INVALID || !DeltaWeightsLSTM.BufferRead() || !DeltaWeightsLSTM.Save(file_handle)) return false; if(CheckPointer(Memory) == POINTER_INVALID || !Memory.BufferRead() || !Memory.Save(file_handle)) return false; //--- return true; } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ bool CNeuronLSTMOCL::Load(const int file_handle) { if(!CNeuronBaseOCL::Load(file_handle)) return false; int savedTag = FileReadInteger(file_handle, INT_VALUE); if(savedTag != LSTM_SEQ_SAVE_TAG) { //--- Pre-sequence .nnw. Its LSTM weight block is shaped for the whole flattened input as one //--- timestep and cannot be reinterpreted; refuse so CNet::Load fails cleanly and the caller //--- rebuilds a fresh topology, rather than reading a differently-shaped buffer and every //--- subsequent layer at the wrong offset. printf("CNeuronLSTMOCL::Load: this model predates the sequence-LSTM rewrite (found %d, expected %d) - it must be retrained. Delete its .nnw (and _shadow.nnw) to start clean.", savedTag, LSTM_SEQ_SAVE_TAG); return false; } m_iInputs = FileReadInteger(file_handle, INT_VALUE); m_iStepInputs = FileReadInteger(file_handle, INT_VALUE); m_iSteps = (m_iStepInputs > 0 && m_iInputs > 0 && (m_iInputs % m_iStepInputs) == 0) ? m_iInputs / m_iStepInputs : -1; uint H = (uint)Neurons(); //--- scratch buffers (not persisted) were sized for the placeholder Init() unit //--- count; resize them now that the real neuron count is known. if(CheckPointer(Concatenated) == POINTER_INVALID) { Concatenated = new CBufferDouble(); if(CheckPointer(Concatenated) == POINTER_INVALID) return false; } Concatenated.BufferFree(); if(!Concatenated.BufferInit(4 * H, 0)) return false; if(!BackendBufferCreate(Concatenated)) return false; if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) { ConcatenatedGradient = new CBufferDouble(); if(CheckPointer(ConcatenatedGradient) == POINTER_INVALID) return false; } ConcatenatedGradient.BufferFree(); if(!ConcatenatedGradient.BufferInit(4 * H, 0)) return false; if(!BackendBufferCreate(ConcatenatedGradient)) return false; if(CheckPointer(HiddenCache) == POINTER_INVALID) { HiddenCache = new CBufferDouble(); if(CheckPointer(HiddenCache) == POINTER_INVALID) return false; } HiddenCache.BufferFree(); if(!HiddenCache.BufferInit(H, 0)) return false; if(!BackendBufferCreate(HiddenCache)) return false; //--- Memory (the c_prev cell state) is allocated HERE, before the m_iInputs<=0 early return below, //--- and NOT only alongside the weight buffers further down. Both Init() overloads allocate it //--- unconditionally; Load() used to allocate it only on the m_iInputs>0 path and SetInputs() never //--- allocates it at all. A net saved before its first feedForward (Save writes m_iInputs=-1 and //--- omits every LSTM buffer - e.g. weights-reset then detach, which is exactly what a "reset and //--- restart" click produces) therefore came back from Load with Memory==NULL. The lazy SetInputs() //--- on the next feedForward rebuilt the weights but not Memory, so LSTMState() got a dead buffer and //--- every forward AND backward pass failed - the layer computed nothing, the head sat at a constant //--- 1.0 for all three classes, and the model was pinned to Neutral forever while the journal filled //--- with "... LSTM feedForward failed" (that line read "Error of execution DirectML LSTM feedForward" //--- until 2026-08-02, when it started naming the backend that actually failed rather than always //--- saying DirectML). If m_iInputs>0 the Memory.Load() further //--- down simply overwrites what we allocate here. if(CheckPointer(Memory) == POINTER_INVALID) { Memory = new CBufferDouble(); if(CheckPointer(Memory) == POINTER_INVALID) return false; } Memory.BufferFree(); if(!Memory.BufferInit(2 * H, 0)) return false; if(!BackendBufferCreate(Memory)) return false; //--- if(m_iInputs <= 0) return true; //--- In sequence mode the gates read ONE timestep, so the weight block is sized on the per-step width. int total = (int)(4 * H * (H + (IsSequenceMode() ? m_iStepInputs : m_iInputs) + 1)); //--- if(CheckPointer(WeightsLSTM) == POINTER_INVALID) { WeightsLSTM = new CBufferDouble(); if(CheckPointer(WeightsLSTM) == POINTER_INVALID) return false; } if(WeightsLSTM.GetIndex() >= 0) WeightsLSTM.BufferFree(); if(!WeightsLSTM.Load(file_handle)) return false; if(!BackendBufferCreate(WeightsLSTM)) return false; //--- if(CheckPointer(FirstMomentumLSTM) == POINTER_INVALID) { FirstMomentumLSTM = new CBufferDouble(); if(CheckPointer(FirstMomentumLSTM) == POINTER_INVALID) return false; } if(FirstMomentumLSTM.GetIndex() >= 0) FirstMomentumLSTM.BufferFree(); if(!FirstMomentumLSTM.Load(file_handle)) return false; if(!BackendBufferCreate(FirstMomentumLSTM)) return false; //--- if(CheckPointer(SecondMomentumLSTM) == POINTER_INVALID) { SecondMomentumLSTM = new CBufferDouble(); if(CheckPointer(SecondMomentumLSTM) == POINTER_INVALID) return false; } if(SecondMomentumLSTM.GetIndex() >= 0) SecondMomentumLSTM.BufferFree(); if(!SecondMomentumLSTM.Load(file_handle)) return false; if(!BackendBufferCreate(SecondMomentumLSTM)) return false; //--- if(CheckPointer(DeltaWeightsLSTM) == POINTER_INVALID) { DeltaWeightsLSTM = new CBufferDouble(); if(CheckPointer(DeltaWeightsLSTM) == POINTER_INVALID) return false; } if(DeltaWeightsLSTM.GetIndex() >= 0) DeltaWeightsLSTM.BufferFree(); if(!DeltaWeightsLSTM.Load(file_handle)) return false; if(!BackendBufferCreate(DeltaWeightsLSTM)) return false; //--- if(CheckPointer(Memory) == POINTER_INVALID) { Memory = new CBufferDouble(); if(CheckPointer(Memory) == POINTER_INVALID) return false; } if(Memory.GetIndex() >= 0) Memory.BufferFree(); if(!Memory.Load(file_handle)) return false; if(!BackendBufferCreate(Memory)) return false; //--- if(CheckPointer(WeightsGradient) == POINTER_INVALID) { WeightsGradient = new CBufferDouble(); if(CheckPointer(WeightsGradient) == POINTER_INVALID) return false; } if(!WeightsGradient.BufferInit(total, 0)) return false; if(!BackendBufferCreate(WeightsGradient)) return false; //--- Per-timestep BPTT caches are scratch and never persisted - rebuild them from the loaded shape. if(!AllocateSequenceCaches()) return false; //--- return true; } #endif