Warrior_EA/AI/Impl/NetWeights.mqh

711 lines
33 KiB
MQL5
Raw Permalink Normal View History

//+------------------------------------------------------------------+
//| NetWeights.mqh |
//| |
//| CNet weight operations: EMA blend, in-memory snapshot/restore, |
//| per-layer learning report, topology contract checks. |
//| |
//| Included from AI\Network.mqh AFTER every class declaration - |
//| bodies only, no declarations. Relocation is behaviour-neutral by |
//| construction: nothing here is reachable until Network.mqh ends. |
//+------------------------------------------------------------------+
#ifndef WARRIOR_AI_IMPL_NETWEIGHTS_MQH
#define WARRIOR_AI_IMPL_NETWEIGHTS_MQH
//+------------------------------------------------------------------+
//| See this method's declaration comment for the EMA shadow-weight |
//| deployment rationale. |
//+------------------------------------------------------------------+
bool CNet::BlendWeightsFrom(CNet &live, double tau)
{
if(CheckPointer(layers) == POINTER_INVALID || CheckPointer(live.layers) == POINTER_INVALID)
return false;
//--- Skip accounting. Every branch below blends only when BOTH sides hand back a weight block, and
//--- silently does nothing otherwise - deliberate, so a partial topology mismatch degrades instead of
//--- corrupting unrelated layers. But "silently" made a real defect invisible for 343 eras on SP500 H1
//--- (2026-07-31): the HYBRID shadow's LSTM layer had never run a forward pass, so its WeightsLSTM was
//--- still NULL, getWeightsLSTM() returned 0, and the blend skipped a 24,704-weight layer on EVERY era
//--- while reporting success. It showed up only as a shadow .nnw 791,120 bytes smaller than its live
//--- net - exactly the LSTM weight block and its Adam moments - which nothing was watching. A skip is
//--- never normal on a shadow cloned from live, so say so ONCE per net rather than never.
int skipped = 0;
int skippedLayer = -1, skippedType = 0;
int layerTotal = MathMin(layers.Total(), live.layers.Total());
for(int l = 0; l < layerTotal; l++)
{
CLayer *shadowLayer = layers.At(l);
CLayer *liveLayer = live.layers.At(l);
if(CheckPointer(shadowLayer) == POINTER_INVALID || CheckPointer(liveLayer) == POINTER_INVALID)
continue;
int neuronTotal = MathMin(shadowLayer.Total(), liveLayer.Total());
for(int n = 0; n < neuronTotal; n++)
{
//--- Type() first, through the true common base (CObject) - NOT through a CNeuronBaseOCL*-typed
//--- pointer. On the plain-CPU tier (no OpenCL/DirectML - the only option on Marketplace, which
//--- forbids DLLs) CNet::CNet() builds legacy CNeuronBase-hierarchy neurons (CNeuron/CNeuronConv/
//--- CNeuronPool/CNeuronLSTM), an UNRELATED class hierarchy from CNeuronBaseOCL/CNeuronConvOCL/
//--- CNeuronLSTMOCL. Assigning one of those objects to a CNeuronBaseOCL* and calling its virtual
//--- methods (as this used to do unconditionally) reads the wrong vtable/member layout - undefined
//--- behaviour, not just a silent no-op, every single online-learning step on that tier.
CObject *shadowObj = shadowLayer.At(n);
CObject *liveObj = liveLayer.At(n);
if(CheckPointer(shadowObj) == POINTER_INVALID || CheckPointer(liveObj) == POINTER_INVALID)
continue;
if(shadowObj.Type() != liveObj.Type())
continue;
double shadowW[], liveW[];
switch(shadowObj.Type())
{
case defNeuronBaseOCL:
{
CNeuronBaseOCL *shadowNeuron = shadowObj;
CNeuronBaseOCL *liveNeuron = liveObj;
int gotShadow = shadowNeuron.getWeights(shadowW);
int gotLive = liveNeuron.getWeights(liveW);
if(gotShadow > 0 && gotLive > 0)
{
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
for(int wi = 0; wi < wt; wi++)
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
shadowNeuron.setWeights(shadowW);
}
else
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
if(gotLive > 0)
{
skipped++;
skippedLayer = l;
skippedType = shadowObj.Type();
}
}
break;
case defNeuronBatchNormOCL:
{
//--- getWeightsBN packs the outgoing dense matrix and gamma/beta/statistics into one
//--- flat array, so the shared blend loop below needs no special case of its own.
CNeuronBatchNormOCL *shadowBN = shadowObj;
CNeuronBatchNormOCL *liveBN = liveObj;
int gotShadow = shadowBN.getWeightsBN(shadowW);
int gotLive = liveBN.getWeightsBN(liveW);
if(gotShadow > 0 && gotLive > 0)
{
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
for(int wi = 0; wi < wt; wi++)
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
shadowBN.setWeightsBN(shadowW);
}
else
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
if(gotLive > 0)
{
skipped++;
skippedLayer = l;
skippedType = shadowObj.Type();
}
}
break;
case defNeuronConvOCL:
{
CNeuronConvOCL *shadowConv = shadowObj;
CNeuronConvOCL *liveConv = liveObj;
int gotShadow = shadowConv.getWeightsConv(shadowW);
int gotLive = liveConv.getWeightsConv(liveW);
if(gotShadow > 0 && gotLive > 0)
{
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
for(int wi = 0; wi < wt; wi++)
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
shadowConv.setWeightsConv(shadowW);
}
else
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
if(gotLive > 0)
{
skipped++;
skippedLayer = l;
skippedType = shadowObj.Type();
}
}
break;
case defNeuronLSTMOCL:
{
CNeuronLSTMOCL *shadowLstm = shadowObj;
CNeuronLSTMOCL *liveLstm = liveObj;
int gotShadow = shadowLstm.getWeightsLSTM(shadowW);
int gotLive = liveLstm.getWeightsLSTM(liveW);
//--- SELF-HEAL, 2026-07-31. EnsureShadowNet() can bootstrap the clone from
//--- RefreshLatestSignal BEFORE the live net has run a single forward pass, and
//--- CNeuronLSTMOCL::Save omits every LSTM buffer for a layer in that state - so the shadow
//--- came back with no WeightsLSTM and, because only the LIVE net ever runs forward, never
//--- got one. The blend then skipped a 24,704-weight layer on EVERY era while returning
//--- true, for a whole 343-era run. It was visible only as a shadow .nnw 791,120 bytes
//--- short of its live net (exactly the LSTM block plus its Adam moments), and the shadow
//--- is what live inference and deployment read.
//--- An EMA whose accumulator does not exist yet must seed at the FIRST observation, not
//--- blend tau of it into freshly randomized weights - hence the outright copy.
if(gotShadow <= 0 && gotLive > 0 && shadowLstm.AdoptShapeFrom(liveLstm))
{
shadowLstm.setWeightsLSTM(liveW);
break;
}
if(gotShadow > 0 && gotLive > 0)
{
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
for(int wi = 0; wi < wt; wi++)
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
shadowLstm.setWeightsLSTM(shadowW);
}
else
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
if(gotLive > 0)
{
skipped++;
skippedLayer = l;
skippedType = shadowObj.Type();
}
}
break;
//--- Plain-CPU tier (no backend at all): dense/conv/pool/LSTM legacy neurons store weights
//--- per-connection (CConnection.weight) rather than in one contiguous buffer - blend those
//--- directly instead of silently dropping the shadow-EMA deployment on this tier.
case defNeuron:
case defNeuronConv:
case defNeuronPool:
case defNeuronLSTM:
{
CNeuronBase *shadowBase = shadowObj;
CNeuronBase *liveBase = liveObj;
CArrayCon *shadowCon = shadowBase.getConnections();
CArrayCon *liveCon = liveBase.getConnections();
if(CheckPointer(shadowCon) == POINTER_INVALID || CheckPointer(liveCon) == POINTER_INVALID)
break;
int ct = MathMin(shadowCon.Total(), liveCon.Total());
for(int ci = 0; ci < ct; ci++)
{
CConnection *sc = shadowCon.At(ci);
CConnection *lc = liveCon.At(ci);
if(CheckPointer(sc) == POINTER_INVALID || CheckPointer(lc) == POINTER_INVALID)
continue;
sc.weight = (1.0 - tau) * sc.weight + tau * lc.weight;
}
}
break;
}
}
}
if(skipped > 0 && !m_blendSkipLogged)
{
m_blendSkipLogged = true;
Print("CNet::BlendWeightsFrom: WARNING - ", skipped, " weight block(s) could not be blended into the EMA shadow ",
"(last: layer ", skippedLayer, ", neuron type ", skippedType, "). The shadow is what live inference and ",
"deployment read, so those layers are NOT tracking the trained model - they keep whatever they were ",
"initialized with. A shadow cloned from the live net should never skip; investigate rather than ignore.");
}
return true;
}
//+------------------------------------------------------------------+
//| Per-layer "is this layer actually learning?" report. |
//| |
//| Returns one token per layer: <type><index>:<|W|>(<relative change |
//| since the previous call>). A layer whose relative change is ~0 era|
//| after era is NOT TRAINING, whatever the loss curve says. |
//| |
//| Why this exists: two separate incidents in this engine presented |
//| identically - a flat metric with the model retreating to the |
//| majority class - and in both the real fault was that one stage |
//| received no usable gradient while every other stage trained |
//| normally. Neither the loss, the accuracy, nor the per-class recall|
//| can distinguish "this layer is frozen" from "this architecture |
//| does not suit the data", and guessing between those two costs a |
//| full retrain per guess. The weight norms distinguish them directly.|
//| See CNet::backProp's layer-1 LSTM special case for the shape the |
//| first such bug took. |
//+------------------------------------------------------------------+
string CNet::LayerLearningReport(void)
{
if(CheckPointer(layers) == POINTER_INVALID)
return "";
int total = layers.Total();
if(ArraySize(m_prevLayerNorm) != total)
{
ArrayResize(m_prevLayerNorm, total);
ArrayInitialize(m_prevLayerNorm, -1.0);
}
//--- One slot per layer index, so At(l) lines up with the loop below without bookkeeping. Layers that
//--- own no weights just keep an empty array.
if(CheckPointer(m_prevLayerWeights) == POINTER_INVALID)
m_prevLayerWeights = new CArrayObj();
if(CheckPointer(m_prevLayerWeights) != POINTER_INVALID)
while(m_prevLayerWeights.Total() < total)
if(!m_prevLayerWeights.Add(new CArrayDouble()))
break;
string report = "";
for(int l = 0; l < total; l++)
{
CLayer *layer = (CLayer*)layers.At(l);
if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0)
continue;
CObject *obj = layer.At(0);
if(CheckPointer(obj) == POINTER_INVALID)
continue;
double w[];
int got = 0;
string tag = "?";
switch(obj.Type())
{
case defNeuronBaseOCL:
{
CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj;
got = n.getWeights(w);
tag = "dense";
break;
}
case defNeuronConvOCL:
{
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
got = c.getWeightsConv(w);
tag = "conv";
break;
}
case defNeuronLSTMOCL:
{
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
got = ls.getWeightsLSTM(w);
tag = "lstm";
break;
}
case defNeuronBatchNormOCL:
{
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
got = bn.getWeightsBN(w);
tag = "bn";
//--- getWeightsBN packs FOUR different kinds of number into one array: the outgoing dense
//--- matrix, the learned gamma/beta, the running mean/variance, and the Adam moment
//--- buffers. A single norm over all of them cannot say which one is moving - and on
//--- 2026-08-02 that ambiguity was the difference between "the weights are diverging"
//--- (an optimizer problem) and "the running variance is tracking activations that grew"
//--- (a scaling problem), which need opposite fixes. PAI reached bn3:465,684 with no way
//--- to tell them apart. So break the block down: the sub-norms cost one pass and turn a
//--- guess into a reading.
int oCount = bn.BatchOptionsTotal();
int wCount = got - oCount;
if(oCount > 0 && wCount >= 0)
{
double nW = 0.0, nGamma = 0.0, nBeta = 0.0, nMean = 0.0, nVar = 0.0, nAdam = 0.0;
for(int i = 0; i < wCount; i++)
nW += w[i] * w[i];
for(int o = 0; o < oCount; o++)
{
double v = w[wCount + o];
switch(o % BN_OPT_STRIDE)
{
case BN_OPT_MEAN: nMean += v * v; break;
case BN_OPT_VAR: nVar += v * v; break;
case BN_OPT_GAMMA: nGamma += v * v; break;
case BN_OPT_BETA: nBeta += v * v; break;
case BN_OPT_MG:
case BN_OPT_MB:
case BN_OPT_VG:
case BN_OPT_VB: nAdam += v * v; break;
default: break; // BN_OPT_NX is a forward-pass scratch value
}
}
report += StringFormat(" bn%d[W %.3g|g %.3g|b %.3g|mean %.3g|var %.3g|adam %.3g]",
l, MathSqrt(nW), MathSqrt(nGamma), MathSqrt(nBeta),
MathSqrt(nMean), MathSqrt(nVar), MathSqrt(nAdam));
}
break;
}
case defNeuronPoolOCL:
//--- No parameters of its own. Named anyway so the report shows the real layer order.
report += " pool" + IntegerToString(l) + ":-";
continue;
default:
continue;
}
if(got <= 0)
{
report += " " + tag + IntegerToString(l) + ":NOWEIGHTS";
continue;
}
double sum = 0.0;
for(int i = 0; i < got; i++)
sum += w[i] * w[i];
double norm = MathSqrt(sum);
double prev = m_prevLayerNorm[l];
m_prevLayerNorm[l] = norm;
//--- |dW|: norm of the elementwise change since the last report. Only meaningful when the previous
//--- vector has the same length (a rebuilt topology invalidates it), hence the size check.
double stepRel = -1.0;
CArrayDouble *prevW = (CheckPointer(m_prevLayerWeights) != POINTER_INVALID && l < m_prevLayerWeights.Total()
? (CArrayDouble*)m_prevLayerWeights.At(l) : NULL);
if(CheckPointer(prevW) != POINTER_INVALID)
{
if(prevW.Total() == got && prev > 0.0)
{
double d2 = 0.0;
for(int i = 0; i < got; i++)
{
double d = w[i] - prevW.At(i);
d2 += d * d;
}
stepRel = MathSqrt(d2) / prev;
}
prevW.Clear();
for(int i = 0; i < got; i++)
prevW.Add(w[i]);
}
if(prev < 0.0)
{
report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) + "(init)";
continue;
}
//--- Relative, so a wide layer and a narrow one are comparable at a glance. Printed as
//--- norm(d|W| / |dW|): the FIRST is how much the length changed, the SECOND how far the vector
//--- actually moved. Decay-only shows the two roughly EQUAL with the norm falling; a learning
//--- layer shows the second clearly larger. See m_prevLayerWeights' declaration comment.
double rel = (prev > 0.0 ? MathAbs(norm - prev) / prev : 0.0);
report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) +
"(" + DoubleToString(100.0 * rel, 3) + "%/" +
(stepRel < 0.0 ? "n/a" : DoubleToString(100.0 * stepRel, 3) + "%") + ")";
}
return report;
}
//+------------------------------------------------------------------+
//| Snapshot every neuron's weights into host memory (see the header |
//| declaration). Layer-major order; one CArrayDouble per neuron. Used|
//| as the mid-run best-era checkpoint - restored by RestoreWeights().|
//+------------------------------------------------------------------+
bool CNet::CaptureWeights(void)
{
if(CheckPointer(layers) == POINTER_INVALID)
return false;
if(CheckPointer(m_weightSnapshot) == POINTER_INVALID)
{
m_weightSnapshot = new CArrayObj();
if(CheckPointer(m_weightSnapshot) == POINTER_INVALID)
return false;
}
m_weightSnapshot.Clear(); // FreeMode deletes the previous snapshot's per-neuron arrays
m_haveWeightSnapshot = false;
for(int l = 0; l < layers.Total(); l++)
{
CLayer *layer = (CLayer*)layers.At(l);
if(CheckPointer(layer) == POINTER_INVALID)
return false;
for(int n = 0; n < layer.Total(); n++)
{
CObject *obj = layer.At(n);
if(CheckPointer(obj) == POINTER_INVALID)
continue;
double w[];
int got = 0;
switch(obj.Type())
{
case defNeuronBaseOCL:
{
CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj;
got = neuron.getWeights(w);
break;
}
case defNeuronBatchNormOCL:
{
//--- Dense matrix + gamma/beta/statistics as one array - see getWeightsBN. Without this
//--- the plateau ladder would restore the weights around this layer while leaving its
//--- own parameters at whatever the diverged era left behind.
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
got = bn.getWeightsBN(w);
break;
}
case defNeuronConvOCL:
{
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
got = c.getWeightsConv(w);
break;
}
case defNeuronLSTMOCL:
{
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
got = ls.getWeightsLSTM(w);
break;
}
case defNeuron:
case defNeuronConv:
case defNeuronPool:
case defNeuronLSTM:
{
CNeuronBase *neuron = (CNeuronBase*)obj;
CArrayCon *conns = neuron.getConnections();
if(CheckPointer(conns) != POINTER_INVALID)
{
got = conns.Total();
ArrayResize(w, got);
for(int i = 0; i < got; i++)
{
CConnection *con = conns.At(i);
w[i] = (CheckPointer(con) != POINTER_INVALID) ? con.weight : 0.0;
}
}
break;
}
default:
break;
}
CArrayDouble *snap = new CArrayDouble();
if(CheckPointer(snap) == POINTER_INVALID)
return false;
if(got > 0)
snap.AssignArray(w);
if(!m_weightSnapshot.Add(snap))
{
delete snap;
return false;
}
}
}
m_haveWeightSnapshot = true;
return true;
}
//+------------------------------------------------------------------+
//| Write the CaptureWeights() snapshot back into the live neurons IN |
//| PLACE (setWeights - no neuron re-creation, no new device tensors).|
//| Aborts (returning false, live weights untouched past that point) |
//| only on a neuron-count mismatch, which never happens within a run |
//| (fixed topology). See the header declaration for the full why. |
//+------------------------------------------------------------------+
bool CNet::RestoreWeights(void)
{
if(!m_haveWeightSnapshot || CheckPointer(m_weightSnapshot) == POINTER_INVALID || CheckPointer(layers) == POINTER_INVALID)
return false;
int idx = 0;
for(int l = 0; l < layers.Total(); l++)
{
CLayer *layer = (CLayer*)layers.At(l);
if(CheckPointer(layer) == POINTER_INVALID)
return false;
for(int n = 0; n < layer.Total(); n++)
{
if(idx >= m_weightSnapshot.Total())
return false; // topology/count mismatch - stop rather than mis-map weights
CObject *obj = layer.At(n);
CArrayDouble *snap = m_weightSnapshot.At(idx);
idx++;
if(CheckPointer(obj) == POINTER_INVALID || CheckPointer(snap) == POINTER_INVALID)
continue;
int cnt = snap.Total();
if(cnt <= 0)
continue; // no weights captured for this neuron (e.g. output layer) - nothing to restore
double w[];
ArrayResize(w, cnt);
for(int i = 0; i < cnt; i++)
w[i] = snap.At(i);
switch(obj.Type())
{
case defNeuronBaseOCL:
{
CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj;
neuron.setWeights(w);
break;
}
case defNeuronBatchNormOCL:
{
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
bn.setWeightsBN(w);
break;
}
case defNeuronConvOCL:
{
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
c.setWeightsConv(w);
break;
}
case defNeuronLSTMOCL:
{
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
ls.setWeightsLSTM(w);
break;
}
case defNeuron:
case defNeuronConv:
case defNeuronPool:
case defNeuronLSTM:
{
CNeuronBase *neuron = (CNeuronBase*)obj;
CArrayCon *conns = neuron.getConnections();
if(CheckPointer(conns) == POINTER_INVALID)
break;
int count = MathMin(cnt, conns.Total());
for(int i = 0; i < count; i++)
{
CConnection *con = conns.At(i);
if(CheckPointer(con) != POINTER_INVALID)
con.weight = w[i];
}
break;
}
default:
break;
}
}
}
return (idx == m_weightSnapshot.Total());
}
//+------------------------------------------------------------------+
//| See this method's declaration comment for the cold-start bias |
//| rationale. The weight block that produces the output layer's |
//| values is stored on the layer BEFORE it (see CNeuronBaseOCL:: |
//| feedForward(CNeuronBaseOCL*) - matrix_w comes from the SOURCE |
//| neuron, laid out as (sourceNeurons+1) values per destination |
//| neuron, the last of which is that neuron's bias term), so this |
//| reaches one layer back from the output layer to edit it. |
//+------------------------------------------------------------------+
bool CNet::SeedOutputLayerBias(const double &biasValues[])
{
int outputs = ArraySize(biasValues);
if(outputs <= 0 || CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2)
return false;
CLayer *sourceLayer = layers.At(layers.Total() - 2);
if(CheckPointer(sourceLayer) == POINTER_INVALID || sourceLayer.Total() <= 0)
return false;
CObject *sourceObj = sourceLayer.At(0);
//--- Batch norm is accepted here as well as a plain dense layer. It is a CNeuronBaseOCL subclass
//--- with the identical (inputs+1)*outputs weight layout - it just happens to be the layer that
//--- carries the head's weight matrix once normalization is enabled (see AI\NeuronBatchNorm.mqh and
//--- BuildFreshTopology's batch-norm insertion). Without this the exact-type check silently failed
//--- and the cold-start output-bias seed stopped being applied for every batch-norm topology.
if(CheckPointer(sourceObj) == POINTER_INVALID ||
(sourceObj.Type() != defNeuronBaseOCL && sourceObj.Type() != defNeuronBatchNormOCL))
return false;
CNeuronBaseOCL *sourceNeuron = (CNeuronBaseOCL*)sourceObj;
if(CheckPointer(sourceNeuron) == POINTER_INVALID)
return false;
int inputs = sourceNeuron.Neurons();
double weights[];
int count = sourceNeuron.getWeights(weights);
if(count != (inputs + 1) * outputs)
return false; // layout doesn't match the assumed dense (source+1)*outputs block - don't guess
for(int i = 0; i < outputs; i++)
weights[(inputs + 1) * i + inputs] = biasValues[i];
return sourceNeuron.setWeights(weights);
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
void CNet::SetBatchNormFrozen(bool frozen)
{
if(CheckPointer(layers) == POINTER_INVALID)
return;
for(int l = 0; l < layers.Total(); l++)
{
CLayer *layer = (CLayer*)layers.At(l);
if(CheckPointer(layer) == POINTER_INVALID)
continue;
for(int n = 0; n < layer.Total(); n++)
{
CObject *obj = layer.At(n);
if(CheckPointer(obj) == POINTER_INVALID || obj.Type() != defNeuronBatchNormOCL)
continue;
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
bn.SetStatsFrozen(frozen);
}
}
}
//+------------------------------------------------------------------+
//| Window of the FIRST convolutional layer in the loaded net, or 0 |
//| when there is none. |
//| |
//| Same problem as EnforceOutputActivation below, different field: a |
//| .nnw stores the window/step each conv layer was BUILT with, so a |
//| model saved before the receptive field changed keeps the old |
//| shape forever and goes on training under an architecture the code |
//| no longer specifies. Unlike the activation this CANNOT be |
//| repaired in place - the weight block is (window+1)*window_out and |
//| a different window is a different tensor - so the caller's only |
//| correct response is to retrain. |
//+------------------------------------------------------------------+
uint CNet::FirstConvWindow(void)
{
if(CheckPointer(layers) == POINTER_INVALID)
return 0;
for(int i = 0; i < layers.Total(); i++)
{
CLayer *layer = layers.At(i);
if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0)
continue;
CObject *obj = layer.At(0);
if(CheckPointer(obj) == POINTER_INVALID)
continue;
if(obj.Type() == defNeuronConvOCL)
{
CNeuronConvOCL *conv = (CNeuronConvOCL*)obj;
return conv.Window();
}
}
return 0;
}
//+------------------------------------------------------------------+
//| See the declaration comment for why a loaded model's activation |
//| cannot be trusted and must be re-asserted from the topology spec. |
//+------------------------------------------------------------------+
bool CNet::EnforceOutputActivation(ENUM_ACTIVATION intended, ENUM_ACTIVATION &previous)
{
previous = intended;
if(CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2)
return false;
CLayer *outputLayer = layers.At(layers.Total() - 1);
if(CheckPointer(outputLayer) == POINTER_INVALID || outputLayer.Total() <= 0)
return false;
CObject *obj = outputLayer.At(0);
if(CheckPointer(obj) == POINTER_INVALID)
return false;
//--- Both neuron models expose the same two accessors (CNeuronBase gained Activation() for exactly
//--- this), so one branch per family is enough - and anything else (conv/pool/LSTM can never be an
//--- output layer in this project's topologies) is left untouched rather than force-cast.
int t = obj.Type();
if(t == defNeuronBaseOCL || t == defNeuronConvOCL || t == defNeuronPoolOCL || t == defNeuronLSTMOCL ||
t == defNeuronBatchNormOCL)
{
CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj;
previous = n.Activation();
if(previous == intended)
return false;
n.SetActivationFunction(intended);
return true;
}
if(t == defNeuron || t == defNeuronConv || t == defNeuronPool || t == defNeuronLSTM)
{
//--- The scalar CPU model stores activation per NEURON, not per layer, so every neuron in the
//--- output layer has to be corrected - not just index 0.
bool repaired = false;
for(int i = 0; i < outputLayer.Total(); i++)
{
CObject *cell = outputLayer.At(i);
if(CheckPointer(cell) == POINTER_INVALID || cell.Type() != defNeuron)
continue;
CNeuronBase *n = (CNeuronBase*)cell;
if(n.Activation() == intended)
continue;
if(!repaired)
previous = n.Activation();
n.SetActivationFunction(intended);
repaired = true;
}
return repaired;
}
return false;
}
#endif