//+------------------------------------------------------------------+ //| NetWeights.mqh | //| | //| CNet weight operations: EMA blend, in-memory snapshot/restore, | //| per-layer learning report, topology contract checks. | //| | //| Included from AI\Network.mqh AFTER every class declaration - | //| bodies only, no declarations. Relocation is behaviour-neutral by | //| construction: nothing here is reachable until Network.mqh ends. | //+------------------------------------------------------------------+ #ifndef WARRIOR_AI_IMPL_NETWEIGHTS_MQH #define WARRIOR_AI_IMPL_NETWEIGHTS_MQH //+------------------------------------------------------------------+ //| See this method's declaration comment for the EMA shadow-weight | //| deployment rationale. | //+------------------------------------------------------------------+ bool CNet::BlendWeightsFrom(CNet &live, double tau) { if(CheckPointer(layers) == POINTER_INVALID || CheckPointer(live.layers) == POINTER_INVALID) return false; //--- Skip accounting. Every branch below blends only when BOTH sides hand back a weight block, and //--- silently does nothing otherwise - deliberate, so a partial topology mismatch degrades instead of //--- corrupting unrelated layers. But "silently" made a real defect invisible for 343 eras on SP500 H1 //--- (2026-07-31): the HYBRID shadow's LSTM layer had never run a forward pass, so its WeightsLSTM was //--- still NULL, getWeightsLSTM() returned 0, and the blend skipped a 24,704-weight layer on EVERY era //--- while reporting success. It showed up only as a shadow .nnw 791,120 bytes smaller than its live //--- net - exactly the LSTM weight block and its Adam moments - which nothing was watching. A skip is //--- never normal on a shadow cloned from live, so say so ONCE per net rather than never. int skipped = 0; int skippedLayer = -1, skippedType = 0; int layerTotal = MathMin(layers.Total(), live.layers.Total()); for(int l = 0; l < layerTotal; l++) { CLayer *shadowLayer = layers.At(l); CLayer *liveLayer = live.layers.At(l); if(CheckPointer(shadowLayer) == POINTER_INVALID || CheckPointer(liveLayer) == POINTER_INVALID) continue; int neuronTotal = MathMin(shadowLayer.Total(), liveLayer.Total()); for(int n = 0; n < neuronTotal; n++) { //--- Type() first, through the true common base (CObject) - NOT through a CNeuronBaseOCL*-typed //--- pointer. On the plain-CPU tier (no OpenCL/DirectML - the only option on Marketplace, which //--- forbids DLLs) CNet::CNet() builds legacy CNeuronBase-hierarchy neurons (CNeuron/CNeuronConv/ //--- CNeuronPool/CNeuronLSTM), an UNRELATED class hierarchy from CNeuronBaseOCL/CNeuronConvOCL/ //--- CNeuronLSTMOCL. Assigning one of those objects to a CNeuronBaseOCL* and calling its virtual //--- methods (as this used to do unconditionally) reads the wrong vtable/member layout - undefined //--- behaviour, not just a silent no-op, every single online-learning step on that tier. CObject *shadowObj = shadowLayer.At(n); CObject *liveObj = liveLayer.At(n); if(CheckPointer(shadowObj) == POINTER_INVALID || CheckPointer(liveObj) == POINTER_INVALID) continue; if(shadowObj.Type() != liveObj.Type()) continue; double shadowW[], liveW[]; switch(shadowObj.Type()) { case defNeuronBaseOCL: { CNeuronBaseOCL *shadowNeuron = shadowObj; CNeuronBaseOCL *liveNeuron = liveObj; int gotShadow = shadowNeuron.getWeights(shadowW); int gotLive = liveNeuron.getWeights(liveW); if(gotShadow > 0 && gotLive > 0) { int wt = MathMin(ArraySize(shadowW), ArraySize(liveW)); for(int wi = 0; wi < wt; wi++) shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi]; shadowNeuron.setWeights(shadowW); } else //--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not. //--- Both-empty is normal and common - a dense layer whose successor owns the weight //--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which //--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that //--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise. if(gotLive > 0) { skipped++; skippedLayer = l; skippedType = shadowObj.Type(); } } break; case defNeuronBatchNormOCL: { //--- getWeightsBN packs the outgoing dense matrix and gamma/beta/statistics into one //--- flat array, so the shared blend loop below needs no special case of its own. CNeuronBatchNormOCL *shadowBN = shadowObj; CNeuronBatchNormOCL *liveBN = liveObj; int gotShadow = shadowBN.getWeightsBN(shadowW); int gotLive = liveBN.getWeightsBN(liveW); if(gotShadow > 0 && gotLive > 0) { int wt = MathMin(ArraySize(shadowW), ArraySize(liveW)); for(int wi = 0; wi < wt; wi++) shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi]; shadowBN.setWeightsBN(shadowW); } else //--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not. //--- Both-empty is normal and common - a dense layer whose successor owns the weight //--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which //--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that //--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise. if(gotLive > 0) { skipped++; skippedLayer = l; skippedType = shadowObj.Type(); } } break; case defNeuronConvOCL: { CNeuronConvOCL *shadowConv = shadowObj; CNeuronConvOCL *liveConv = liveObj; int gotShadow = shadowConv.getWeightsConv(shadowW); int gotLive = liveConv.getWeightsConv(liveW); if(gotShadow > 0 && gotLive > 0) { int wt = MathMin(ArraySize(shadowW), ArraySize(liveW)); for(int wi = 0; wi < wt; wi++) shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi]; shadowConv.setWeightsConv(shadowW); } else //--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not. //--- Both-empty is normal and common - a dense layer whose successor owns the weight //--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which //--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that //--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise. if(gotLive > 0) { skipped++; skippedLayer = l; skippedType = shadowObj.Type(); } } break; case defNeuronLSTMOCL: { CNeuronLSTMOCL *shadowLstm = shadowObj; CNeuronLSTMOCL *liveLstm = liveObj; int gotShadow = shadowLstm.getWeightsLSTM(shadowW); int gotLive = liveLstm.getWeightsLSTM(liveW); //--- SELF-HEAL, 2026-07-31. EnsureShadowNet() can bootstrap the clone from //--- RefreshLatestSignal BEFORE the live net has run a single forward pass, and //--- CNeuronLSTMOCL::Save omits every LSTM buffer for a layer in that state - so the shadow //--- came back with no WeightsLSTM and, because only the LIVE net ever runs forward, never //--- got one. The blend then skipped a 24,704-weight layer on EVERY era while returning //--- true, for a whole 343-era run. It was visible only as a shadow .nnw 791,120 bytes //--- short of its live net (exactly the LSTM block plus its Adam moments), and the shadow //--- is what live inference and deployment read. //--- An EMA whose accumulator does not exist yet must seed at the FIRST observation, not //--- blend tau of it into freshly randomized weights - hence the outright copy. if(gotShadow <= 0 && gotLive > 0 && shadowLstm.AdoptShapeFrom(liveLstm)) { shadowLstm.setWeightsLSTM(liveW); break; } if(gotShadow > 0 && gotLive > 0) { int wt = MathMin(ArraySize(shadowW), ArraySize(liveW)); for(int wi = 0; wi < wt; wi++) shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi]; shadowLstm.setWeightsLSTM(shadowW); } else //--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not. //--- Both-empty is normal and common - a dense layer whose successor owns the weight //--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which //--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that //--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise. if(gotLive > 0) { skipped++; skippedLayer = l; skippedType = shadowObj.Type(); } } break; //--- Plain-CPU tier (no backend at all): dense/conv/pool/LSTM legacy neurons store weights //--- per-connection (CConnection.weight) rather than in one contiguous buffer - blend those //--- directly instead of silently dropping the shadow-EMA deployment on this tier. case defNeuron: case defNeuronConv: case defNeuronPool: case defNeuronLSTM: { CNeuronBase *shadowBase = shadowObj; CNeuronBase *liveBase = liveObj; CArrayCon *shadowCon = shadowBase.getConnections(); CArrayCon *liveCon = liveBase.getConnections(); if(CheckPointer(shadowCon) == POINTER_INVALID || CheckPointer(liveCon) == POINTER_INVALID) break; int ct = MathMin(shadowCon.Total(), liveCon.Total()); for(int ci = 0; ci < ct; ci++) { CConnection *sc = shadowCon.At(ci); CConnection *lc = liveCon.At(ci); if(CheckPointer(sc) == POINTER_INVALID || CheckPointer(lc) == POINTER_INVALID) continue; sc.weight = (1.0 - tau) * sc.weight + tau * lc.weight; } } break; } } } if(skipped > 0 && !m_blendSkipLogged) { m_blendSkipLogged = true; Print("CNet::BlendWeightsFrom: WARNING - ", skipped, " weight block(s) could not be blended into the EMA shadow ", "(last: layer ", skippedLayer, ", neuron type ", skippedType, "). The shadow is what live inference and ", "deployment read, so those layers are NOT tracking the trained model - they keep whatever they were ", "initialized with. A shadow cloned from the live net should never skip; investigate rather than ignore."); } return true; } //+------------------------------------------------------------------+ //| Per-layer "is this layer actually learning?" report. | //| | //| Returns one token per layer: :<|W|>(). A layer whose relative change is ~0 era| //| after era is NOT TRAINING, whatever the loss curve says. | //| | //| Why this exists: two separate incidents in this engine presented | //| identically - a flat metric with the model retreating to the | //| majority class - and in both the real fault was that one stage | //| received no usable gradient while every other stage trained | //| normally. Neither the loss, the accuracy, nor the per-class recall| //| can distinguish "this layer is frozen" from "this architecture | //| does not suit the data", and guessing between those two costs a | //| full retrain per guess. The weight norms distinguish them directly.| //| See CNet::backProp's layer-1 LSTM special case for the shape the | //| first such bug took. | //+------------------------------------------------------------------+ string CNet::LayerLearningReport(void) { if(CheckPointer(layers) == POINTER_INVALID) return ""; int total = layers.Total(); if(ArraySize(m_prevLayerNorm) != total) { ArrayResize(m_prevLayerNorm, total); ArrayInitialize(m_prevLayerNorm, -1.0); } //--- One slot per layer index, so At(l) lines up with the loop below without bookkeeping. Layers that //--- own no weights just keep an empty array. if(CheckPointer(m_prevLayerWeights) == POINTER_INVALID) m_prevLayerWeights = new CArrayObj(); if(CheckPointer(m_prevLayerWeights) != POINTER_INVALID) while(m_prevLayerWeights.Total() < total) if(!m_prevLayerWeights.Add(new CArrayDouble())) break; string report = ""; for(int l = 0; l < total; l++) { CLayer *layer = (CLayer*)layers.At(l); if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0) continue; CObject *obj = layer.At(0); if(CheckPointer(obj) == POINTER_INVALID) continue; double w[]; int got = 0; string tag = "?"; switch(obj.Type()) { case defNeuronBaseOCL: { CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj; got = n.getWeights(w); tag = "dense"; break; } case defNeuronConvOCL: { CNeuronConvOCL *c = (CNeuronConvOCL*)obj; got = c.getWeightsConv(w); tag = "conv"; break; } case defNeuronLSTMOCL: { CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj; got = ls.getWeightsLSTM(w); tag = "lstm"; break; } case defNeuronBatchNormOCL: { CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj; got = bn.getWeightsBN(w); tag = "bn"; //--- getWeightsBN packs FOUR different kinds of number into one array: the outgoing dense //--- matrix, the learned gamma/beta, the running mean/variance, and the Adam moment //--- buffers. A single norm over all of them cannot say which one is moving - and on //--- 2026-08-02 that ambiguity was the difference between "the weights are diverging" //--- (an optimizer problem) and "the running variance is tracking activations that grew" //--- (a scaling problem), which need opposite fixes. PAI reached bn3:465,684 with no way //--- to tell them apart. So break the block down: the sub-norms cost one pass and turn a //--- guess into a reading. int oCount = bn.BatchOptionsTotal(); int wCount = got - oCount; if(oCount > 0 && wCount >= 0) { double nW = 0.0, nGamma = 0.0, nBeta = 0.0, nMean = 0.0, nVar = 0.0, nAdam = 0.0; for(int i = 0; i < wCount; i++) nW += w[i] * w[i]; for(int o = 0; o < oCount; o++) { double v = w[wCount + o]; switch(o % BN_OPT_STRIDE) { case BN_OPT_MEAN: nMean += v * v; break; case BN_OPT_VAR: nVar += v * v; break; case BN_OPT_GAMMA: nGamma += v * v; break; case BN_OPT_BETA: nBeta += v * v; break; case BN_OPT_MG: case BN_OPT_MB: case BN_OPT_VG: case BN_OPT_VB: nAdam += v * v; break; default: break; // BN_OPT_NX is a forward-pass scratch value } } report += StringFormat(" bn%d[W %.3g|g %.3g|b %.3g|mean %.3g|var %.3g|adam %.3g]", l, MathSqrt(nW), MathSqrt(nGamma), MathSqrt(nBeta), MathSqrt(nMean), MathSqrt(nVar), MathSqrt(nAdam)); } break; } case defNeuronPoolOCL: //--- No parameters of its own. Named anyway so the report shows the real layer order. report += " pool" + IntegerToString(l) + ":-"; continue; default: continue; } if(got <= 0) { report += " " + tag + IntegerToString(l) + ":NOWEIGHTS"; continue; } double sum = 0.0; for(int i = 0; i < got; i++) sum += w[i] * w[i]; double norm = MathSqrt(sum); double prev = m_prevLayerNorm[l]; m_prevLayerNorm[l] = norm; //--- |dW|: norm of the elementwise change since the last report. Only meaningful when the previous //--- vector has the same length (a rebuilt topology invalidates it), hence the size check. double stepRel = -1.0; CArrayDouble *prevW = (CheckPointer(m_prevLayerWeights) != POINTER_INVALID && l < m_prevLayerWeights.Total() ? (CArrayDouble*)m_prevLayerWeights.At(l) : NULL); if(CheckPointer(prevW) != POINTER_INVALID) { if(prevW.Total() == got && prev > 0.0) { double d2 = 0.0; for(int i = 0; i < got; i++) { double d = w[i] - prevW.At(i); d2 += d * d; } stepRel = MathSqrt(d2) / prev; } prevW.Clear(); for(int i = 0; i < got; i++) prevW.Add(w[i]); } if(prev < 0.0) { report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) + "(init)"; continue; } //--- Relative, so a wide layer and a narrow one are comparable at a glance. Printed as //--- norm(d|W| / |dW|): the FIRST is how much the length changed, the SECOND how far the vector //--- actually moved. Decay-only shows the two roughly EQUAL with the norm falling; a learning //--- layer shows the second clearly larger. See m_prevLayerWeights' declaration comment. double rel = (prev > 0.0 ? MathAbs(norm - prev) / prev : 0.0); report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) + "(" + DoubleToString(100.0 * rel, 3) + "%/" + (stepRel < 0.0 ? "n/a" : DoubleToString(100.0 * stepRel, 3) + "%") + ")"; } return report; } //+------------------------------------------------------------------+ //| Snapshot every neuron's weights into host memory (see the header | //| declaration). Layer-major order; one CArrayDouble per neuron. Used| //| as the mid-run best-era checkpoint - restored by RestoreWeights().| //+------------------------------------------------------------------+ bool CNet::CaptureWeights(void) { if(CheckPointer(layers) == POINTER_INVALID) return false; if(CheckPointer(m_weightSnapshot) == POINTER_INVALID) { m_weightSnapshot = new CArrayObj(); if(CheckPointer(m_weightSnapshot) == POINTER_INVALID) return false; } m_weightSnapshot.Clear(); // FreeMode deletes the previous snapshot's per-neuron arrays m_haveWeightSnapshot = false; for(int l = 0; l < layers.Total(); l++) { CLayer *layer = (CLayer*)layers.At(l); if(CheckPointer(layer) == POINTER_INVALID) return false; for(int n = 0; n < layer.Total(); n++) { CObject *obj = layer.At(n); if(CheckPointer(obj) == POINTER_INVALID) continue; double w[]; int got = 0; switch(obj.Type()) { case defNeuronBaseOCL: { CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj; got = neuron.getWeights(w); break; } case defNeuronBatchNormOCL: { //--- Dense matrix + gamma/beta/statistics as one array - see getWeightsBN. Without this //--- the plateau ladder would restore the weights around this layer while leaving its //--- own parameters at whatever the diverged era left behind. CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj; got = bn.getWeightsBN(w); break; } case defNeuronConvOCL: { CNeuronConvOCL *c = (CNeuronConvOCL*)obj; got = c.getWeightsConv(w); break; } case defNeuronLSTMOCL: { CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj; got = ls.getWeightsLSTM(w); break; } case defNeuron: case defNeuronConv: case defNeuronPool: case defNeuronLSTM: { CNeuronBase *neuron = (CNeuronBase*)obj; CArrayCon *conns = neuron.getConnections(); if(CheckPointer(conns) != POINTER_INVALID) { got = conns.Total(); ArrayResize(w, got); for(int i = 0; i < got; i++) { CConnection *con = conns.At(i); w[i] = (CheckPointer(con) != POINTER_INVALID) ? con.weight : 0.0; } } break; } default: break; } CArrayDouble *snap = new CArrayDouble(); if(CheckPointer(snap) == POINTER_INVALID) return false; if(got > 0) snap.AssignArray(w); if(!m_weightSnapshot.Add(snap)) { delete snap; return false; } } } m_haveWeightSnapshot = true; return true; } //+------------------------------------------------------------------+ //| Write the CaptureWeights() snapshot back into the live neurons IN | //| PLACE (setWeights - no neuron re-creation, no new device tensors).| //| Aborts (returning false, live weights untouched past that point) | //| only on a neuron-count mismatch, which never happens within a run | //| (fixed topology). See the header declaration for the full why. | //+------------------------------------------------------------------+ bool CNet::RestoreWeights(void) { if(!m_haveWeightSnapshot || CheckPointer(m_weightSnapshot) == POINTER_INVALID || CheckPointer(layers) == POINTER_INVALID) return false; int idx = 0; for(int l = 0; l < layers.Total(); l++) { CLayer *layer = (CLayer*)layers.At(l); if(CheckPointer(layer) == POINTER_INVALID) return false; for(int n = 0; n < layer.Total(); n++) { if(idx >= m_weightSnapshot.Total()) return false; // topology/count mismatch - stop rather than mis-map weights CObject *obj = layer.At(n); CArrayDouble *snap = m_weightSnapshot.At(idx); idx++; if(CheckPointer(obj) == POINTER_INVALID || CheckPointer(snap) == POINTER_INVALID) continue; int cnt = snap.Total(); if(cnt <= 0) continue; // no weights captured for this neuron (e.g. output layer) - nothing to restore double w[]; ArrayResize(w, cnt); for(int i = 0; i < cnt; i++) w[i] = snap.At(i); switch(obj.Type()) { case defNeuronBaseOCL: { CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj; neuron.setWeights(w); break; } case defNeuronBatchNormOCL: { CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj; bn.setWeightsBN(w); break; } case defNeuronConvOCL: { CNeuronConvOCL *c = (CNeuronConvOCL*)obj; c.setWeightsConv(w); break; } case defNeuronLSTMOCL: { CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj; ls.setWeightsLSTM(w); break; } case defNeuron: case defNeuronConv: case defNeuronPool: case defNeuronLSTM: { CNeuronBase *neuron = (CNeuronBase*)obj; CArrayCon *conns = neuron.getConnections(); if(CheckPointer(conns) == POINTER_INVALID) break; int count = MathMin(cnt, conns.Total()); for(int i = 0; i < count; i++) { CConnection *con = conns.At(i); if(CheckPointer(con) != POINTER_INVALID) con.weight = w[i]; } break; } default: break; } } } return (idx == m_weightSnapshot.Total()); } //+------------------------------------------------------------------+ //| See this method's declaration comment for the cold-start bias | //| rationale. The weight block that produces the output layer's | //| values is stored on the layer BEFORE it (see CNeuronBaseOCL:: | //| feedForward(CNeuronBaseOCL*) - matrix_w comes from the SOURCE | //| neuron, laid out as (sourceNeurons+1) values per destination | //| neuron, the last of which is that neuron's bias term), so this | //| reaches one layer back from the output layer to edit it. | //+------------------------------------------------------------------+ bool CNet::SeedOutputLayerBias(const double &biasValues[]) { int outputs = ArraySize(biasValues); if(outputs <= 0 || CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2) return false; CLayer *sourceLayer = layers.At(layers.Total() - 2); if(CheckPointer(sourceLayer) == POINTER_INVALID || sourceLayer.Total() <= 0) return false; CObject *sourceObj = sourceLayer.At(0); //--- Batch norm is accepted here as well as a plain dense layer. It is a CNeuronBaseOCL subclass //--- with the identical (inputs+1)*outputs weight layout - it just happens to be the layer that //--- carries the head's weight matrix once normalization is enabled (see AI\NeuronBatchNorm.mqh and //--- BuildFreshTopology's batch-norm insertion). Without this the exact-type check silently failed //--- and the cold-start output-bias seed stopped being applied for every batch-norm topology. if(CheckPointer(sourceObj) == POINTER_INVALID || (sourceObj.Type() != defNeuronBaseOCL && sourceObj.Type() != defNeuronBatchNormOCL)) return false; CNeuronBaseOCL *sourceNeuron = (CNeuronBaseOCL*)sourceObj; if(CheckPointer(sourceNeuron) == POINTER_INVALID) return false; int inputs = sourceNeuron.Neurons(); double weights[]; int count = sourceNeuron.getWeights(weights); if(count != (inputs + 1) * outputs) return false; // layout doesn't match the assumed dense (source+1)*outputs block - don't guess for(int i = 0; i < outputs; i++) weights[(inputs + 1) * i + inputs] = biasValues[i]; return sourceNeuron.setWeights(weights); } //+------------------------------------------------------------------+ //| | //+------------------------------------------------------------------+ void CNet::SetBatchNormFrozen(bool frozen) { if(CheckPointer(layers) == POINTER_INVALID) return; for(int l = 0; l < layers.Total(); l++) { CLayer *layer = (CLayer*)layers.At(l); if(CheckPointer(layer) == POINTER_INVALID) continue; for(int n = 0; n < layer.Total(); n++) { CObject *obj = layer.At(n); if(CheckPointer(obj) == POINTER_INVALID || obj.Type() != defNeuronBatchNormOCL) continue; CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj; bn.SetStatsFrozen(frozen); } } } //+------------------------------------------------------------------+ //| Window of the FIRST convolutional layer in the loaded net, or 0 | //| when there is none. | //| | //| Same problem as EnforceOutputActivation below, different field: a | //| .nnw stores the window/step each conv layer was BUILT with, so a | //| model saved before the receptive field changed keeps the old | //| shape forever and goes on training under an architecture the code | //| no longer specifies. Unlike the activation this CANNOT be | //| repaired in place - the weight block is (window+1)*window_out and | //| a different window is a different tensor - so the caller's only | //| correct response is to retrain. | //+------------------------------------------------------------------+ uint CNet::FirstConvWindow(void) { if(CheckPointer(layers) == POINTER_INVALID) return 0; for(int i = 0; i < layers.Total(); i++) { CLayer *layer = layers.At(i); if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0) continue; CObject *obj = layer.At(0); if(CheckPointer(obj) == POINTER_INVALID) continue; if(obj.Type() == defNeuronConvOCL) { CNeuronConvOCL *conv = (CNeuronConvOCL*)obj; return conv.Window(); } } return 0; } //+------------------------------------------------------------------+ //| See the declaration comment for why a loaded model's activation | //| cannot be trusted and must be re-asserted from the topology spec. | //+------------------------------------------------------------------+ bool CNet::EnforceOutputActivation(ENUM_ACTIVATION intended, ENUM_ACTIVATION &previous) { previous = intended; if(CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2) return false; CLayer *outputLayer = layers.At(layers.Total() - 1); if(CheckPointer(outputLayer) == POINTER_INVALID || outputLayer.Total() <= 0) return false; CObject *obj = outputLayer.At(0); if(CheckPointer(obj) == POINTER_INVALID) return false; //--- Both neuron models expose the same two accessors (CNeuronBase gained Activation() for exactly //--- this), so one branch per family is enough - and anything else (conv/pool/LSTM can never be an //--- output layer in this project's topologies) is left untouched rather than force-cast. int t = obj.Type(); if(t == defNeuronBaseOCL || t == defNeuronConvOCL || t == defNeuronPoolOCL || t == defNeuronLSTMOCL || t == defNeuronBatchNormOCL) { CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj; previous = n.Activation(); if(previous == intended) return false; n.SetActivationFunction(intended); return true; } if(t == defNeuron || t == defNeuronConv || t == defNeuronPool || t == defNeuronLSTM) { //--- The scalar CPU model stores activation per NEURON, not per layer, so every neuron in the //--- output layer has to be corrected - not just index 0. bool repaired = false; for(int i = 0; i < outputLayer.Total(); i++) { CObject *cell = outputLayer.At(i); if(CheckPointer(cell) == POINTER_INVALID || cell.Type() != defNeuron) continue; CNeuronBase *n = (CNeuronBase*)cell; if(n.Activation() == intended) continue; if(!repaired) previous = n.Activation(); n.SetActivationFunction(intended); repaired = true; } return repaired; } return false; } #endif