2026-08-01 11:27:28 -04:00
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| NetWeights.mqh |
|
|
|
|
|
//| |
|
|
|
|
|
//| CNet weight operations: EMA blend, in-memory snapshot/restore, |
|
|
|
|
|
//| per-layer learning report, topology contract checks. |
|
|
|
|
|
//| |
|
|
|
|
|
//| Included from AI\Network.mqh AFTER every class declaration - |
|
|
|
|
|
//| bodies only, no declarations. Relocation is behaviour-neutral by |
|
|
|
|
|
//| construction: nothing here is reachable until Network.mqh ends. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
#ifndef WARRIOR_AI_IMPL_NETWEIGHTS_MQH
|
|
|
|
|
#define WARRIOR_AI_IMPL_NETWEIGHTS_MQH
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| See this method's declaration comment for the EMA shadow-weight |
|
|
|
|
|
//| deployment rationale. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
bool CNet::BlendWeightsFrom(CNet &live, double tau)
|
|
|
|
|
{
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID || CheckPointer(live.layers) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
//--- Skip accounting. Every branch below blends only when BOTH sides hand back a weight block, and
|
|
|
|
|
//--- silently does nothing otherwise - deliberate, so a partial topology mismatch degrades instead of
|
|
|
|
|
//--- corrupting unrelated layers. But "silently" made a real defect invisible for 343 eras on SP500 H1
|
|
|
|
|
//--- (2026-07-31): the HYBRID shadow's LSTM layer had never run a forward pass, so its WeightsLSTM was
|
|
|
|
|
//--- still NULL, getWeightsLSTM() returned 0, and the blend skipped a 24,704-weight layer on EVERY era
|
|
|
|
|
//--- while reporting success. It showed up only as a shadow .nnw 791,120 bytes smaller than its live
|
|
|
|
|
//--- net - exactly the LSTM weight block and its Adam moments - which nothing was watching. A skip is
|
|
|
|
|
//--- never normal on a shadow cloned from live, so say so ONCE per net rather than never.
|
|
|
|
|
int skipped = 0;
|
|
|
|
|
int skippedLayer = -1, skippedType = 0;
|
|
|
|
|
int layerTotal = MathMin(layers.Total(), live.layers.Total());
|
|
|
|
|
for(int l = 0; l < layerTotal; l++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *shadowLayer = layers.At(l);
|
|
|
|
|
CLayer *liveLayer = live.layers.At(l);
|
|
|
|
|
if(CheckPointer(shadowLayer) == POINTER_INVALID || CheckPointer(liveLayer) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
int neuronTotal = MathMin(shadowLayer.Total(), liveLayer.Total());
|
|
|
|
|
for(int n = 0; n < neuronTotal; n++)
|
|
|
|
|
{
|
|
|
|
|
//--- Type() first, through the true common base (CObject) - NOT through a CNeuronBaseOCL*-typed
|
|
|
|
|
//--- pointer. On the plain-CPU tier (no OpenCL/DirectML - the only option on Marketplace, which
|
|
|
|
|
//--- forbids DLLs) CNet::CNet() builds legacy CNeuronBase-hierarchy neurons (CNeuron/CNeuronConv/
|
|
|
|
|
//--- CNeuronPool/CNeuronLSTM), an UNRELATED class hierarchy from CNeuronBaseOCL/CNeuronConvOCL/
|
|
|
|
|
//--- CNeuronLSTMOCL. Assigning one of those objects to a CNeuronBaseOCL* and calling its virtual
|
|
|
|
|
//--- methods (as this used to do unconditionally) reads the wrong vtable/member layout - undefined
|
|
|
|
|
//--- behaviour, not just a silent no-op, every single online-learning step on that tier.
|
|
|
|
|
CObject *shadowObj = shadowLayer.At(n);
|
|
|
|
|
CObject *liveObj = liveLayer.At(n);
|
|
|
|
|
if(CheckPointer(shadowObj) == POINTER_INVALID || CheckPointer(liveObj) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
if(shadowObj.Type() != liveObj.Type())
|
|
|
|
|
continue;
|
|
|
|
|
double shadowW[], liveW[];
|
|
|
|
|
switch(shadowObj.Type())
|
|
|
|
|
{
|
|
|
|
|
case defNeuronBaseOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBaseOCL *shadowNeuron = shadowObj;
|
|
|
|
|
CNeuronBaseOCL *liveNeuron = liveObj;
|
|
|
|
|
int gotShadow = shadowNeuron.getWeights(shadowW);
|
|
|
|
|
int gotLive = liveNeuron.getWeights(liveW);
|
|
|
|
|
if(gotShadow > 0 && gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
|
|
|
|
|
for(int wi = 0; wi < wt; wi++)
|
|
|
|
|
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
|
|
|
|
|
shadowNeuron.setWeights(shadowW);
|
|
|
|
|
}
|
|
|
|
|
else
|
|
|
|
|
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
|
|
|
|
|
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
|
|
|
|
|
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
|
|
|
|
|
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
|
|
|
|
|
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
|
|
|
|
|
if(gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
skipped++;
|
|
|
|
|
skippedLayer = l;
|
|
|
|
|
skippedType = shadowObj.Type();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
case defNeuronBatchNormOCL:
|
|
|
|
|
{
|
|
|
|
|
//--- getWeightsBN packs the outgoing dense matrix and gamma/beta/statistics into one
|
|
|
|
|
//--- flat array, so the shared blend loop below needs no special case of its own.
|
|
|
|
|
CNeuronBatchNormOCL *shadowBN = shadowObj;
|
|
|
|
|
CNeuronBatchNormOCL *liveBN = liveObj;
|
|
|
|
|
int gotShadow = shadowBN.getWeightsBN(shadowW);
|
|
|
|
|
int gotLive = liveBN.getWeightsBN(liveW);
|
|
|
|
|
if(gotShadow > 0 && gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
|
|
|
|
|
for(int wi = 0; wi < wt; wi++)
|
|
|
|
|
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
|
|
|
|
|
shadowBN.setWeightsBN(shadowW);
|
|
|
|
|
}
|
|
|
|
|
else
|
|
|
|
|
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
|
|
|
|
|
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
|
|
|
|
|
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
|
|
|
|
|
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
|
|
|
|
|
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
|
|
|
|
|
if(gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
skipped++;
|
|
|
|
|
skippedLayer = l;
|
|
|
|
|
skippedType = shadowObj.Type();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
case defNeuronConvOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronConvOCL *shadowConv = shadowObj;
|
|
|
|
|
CNeuronConvOCL *liveConv = liveObj;
|
|
|
|
|
int gotShadow = shadowConv.getWeightsConv(shadowW);
|
|
|
|
|
int gotLive = liveConv.getWeightsConv(liveW);
|
|
|
|
|
if(gotShadow > 0 && gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
|
|
|
|
|
for(int wi = 0; wi < wt; wi++)
|
|
|
|
|
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
|
|
|
|
|
shadowConv.setWeightsConv(shadowW);
|
|
|
|
|
}
|
|
|
|
|
else
|
|
|
|
|
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
|
|
|
|
|
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
|
|
|
|
|
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
|
|
|
|
|
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
|
|
|
|
|
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
|
|
|
|
|
if(gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
skipped++;
|
|
|
|
|
skippedLayer = l;
|
|
|
|
|
skippedType = shadowObj.Type();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
case defNeuronLSTMOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronLSTMOCL *shadowLstm = shadowObj;
|
|
|
|
|
CNeuronLSTMOCL *liveLstm = liveObj;
|
|
|
|
|
int gotShadow = shadowLstm.getWeightsLSTM(shadowW);
|
|
|
|
|
int gotLive = liveLstm.getWeightsLSTM(liveW);
|
|
|
|
|
//--- SELF-HEAL, 2026-07-31. EnsureShadowNet() can bootstrap the clone from
|
|
|
|
|
//--- RefreshLatestSignal BEFORE the live net has run a single forward pass, and
|
|
|
|
|
//--- CNeuronLSTMOCL::Save omits every LSTM buffer for a layer in that state - so the shadow
|
|
|
|
|
//--- came back with no WeightsLSTM and, because only the LIVE net ever runs forward, never
|
|
|
|
|
//--- got one. The blend then skipped a 24,704-weight layer on EVERY era while returning
|
|
|
|
|
//--- true, for a whole 343-era run. It was visible only as a shadow .nnw 791,120 bytes
|
|
|
|
|
//--- short of its live net (exactly the LSTM block plus its Adam moments), and the shadow
|
|
|
|
|
//--- is what live inference and deployment read.
|
|
|
|
|
//--- An EMA whose accumulator does not exist yet must seed at the FIRST observation, not
|
|
|
|
|
//--- blend tau of it into freshly randomized weights - hence the outright copy.
|
|
|
|
|
if(gotShadow <= 0 && gotLive > 0 && shadowLstm.AdoptShapeFrom(liveLstm))
|
|
|
|
|
{
|
|
|
|
|
shadowLstm.setWeightsLSTM(liveW);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
if(gotShadow > 0 && gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
int wt = MathMin(ArraySize(shadowW), ArraySize(liveW));
|
|
|
|
|
for(int wi = 0; wi < wt; wi++)
|
|
|
|
|
shadowW[wi] = (1.0 - tau) * shadowW[wi] + tau * liveW[wi];
|
|
|
|
|
shadowLstm.setWeightsLSTM(shadowW);
|
|
|
|
|
}
|
|
|
|
|
else
|
|
|
|
|
//--- Only an ASYMMETRY is a defect: live has a weight block, the shadow does not.
|
|
|
|
|
//--- Both-empty is normal and common - a dense layer whose successor owns the weight
|
|
|
|
|
//--- matrix (see CNeuronBaseOCL::Init's numOutputs>0 guard) legitimately has none, which
|
|
|
|
|
//--- is also why LayerLearningReport prints NOWEIGHTS for those layers. Warning on that
|
|
|
|
|
//--- fired 5-6 times per net on every topology (2026-07-31) and was pure noise.
|
|
|
|
|
if(gotLive > 0)
|
|
|
|
|
{
|
|
|
|
|
skipped++;
|
|
|
|
|
skippedLayer = l;
|
|
|
|
|
skippedType = shadowObj.Type();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
//--- Plain-CPU tier (no backend at all): dense/conv/pool/LSTM legacy neurons store weights
|
|
|
|
|
//--- per-connection (CConnection.weight) rather than in one contiguous buffer - blend those
|
|
|
|
|
//--- directly instead of silently dropping the shadow-EMA deployment on this tier.
|
|
|
|
|
case defNeuron:
|
|
|
|
|
case defNeuronConv:
|
|
|
|
|
case defNeuronPool:
|
|
|
|
|
case defNeuronLSTM:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBase *shadowBase = shadowObj;
|
|
|
|
|
CNeuronBase *liveBase = liveObj;
|
|
|
|
|
CArrayCon *shadowCon = shadowBase.getConnections();
|
|
|
|
|
CArrayCon *liveCon = liveBase.getConnections();
|
|
|
|
|
if(CheckPointer(shadowCon) == POINTER_INVALID || CheckPointer(liveCon) == POINTER_INVALID)
|
|
|
|
|
break;
|
|
|
|
|
int ct = MathMin(shadowCon.Total(), liveCon.Total());
|
|
|
|
|
for(int ci = 0; ci < ct; ci++)
|
|
|
|
|
{
|
|
|
|
|
CConnection *sc = shadowCon.At(ci);
|
|
|
|
|
CConnection *lc = liveCon.At(ci);
|
|
|
|
|
if(CheckPointer(sc) == POINTER_INVALID || CheckPointer(lc) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
sc.weight = (1.0 - tau) * sc.weight + tau * lc.weight;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if(skipped > 0 && !m_blendSkipLogged)
|
|
|
|
|
{
|
|
|
|
|
m_blendSkipLogged = true;
|
|
|
|
|
Print("CNet::BlendWeightsFrom: WARNING - ", skipped, " weight block(s) could not be blended into the EMA shadow ",
|
|
|
|
|
"(last: layer ", skippedLayer, ", neuron type ", skippedType, "). The shadow is what live inference and ",
|
|
|
|
|
"deployment read, so those layers are NOT tracking the trained model - they keep whatever they were ",
|
|
|
|
|
"initialized with. A shadow cloned from the live net should never skip; investigate rather than ignore.");
|
|
|
|
|
}
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| Per-layer "is this layer actually learning?" report. |
|
|
|
|
|
//| |
|
|
|
|
|
//| Returns one token per layer: <type><index>:<|W|>(<relative change |
|
|
|
|
|
//| since the previous call>). A layer whose relative change is ~0 era|
|
|
|
|
|
//| after era is NOT TRAINING, whatever the loss curve says. |
|
|
|
|
|
//| |
|
|
|
|
|
//| Why this exists: two separate incidents in this engine presented |
|
|
|
|
|
//| identically - a flat metric with the model retreating to the |
|
|
|
|
|
//| majority class - and in both the real fault was that one stage |
|
|
|
|
|
//| received no usable gradient while every other stage trained |
|
|
|
|
|
//| normally. Neither the loss, the accuracy, nor the per-class recall|
|
|
|
|
|
//| can distinguish "this layer is frozen" from "this architecture |
|
|
|
|
|
//| does not suit the data", and guessing between those two costs a |
|
|
|
|
|
//| full retrain per guess. The weight norms distinguish them directly.|
|
|
|
|
|
//| See CNet::backProp's layer-1 LSTM special case for the shape the |
|
|
|
|
|
//| first such bug took. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
string CNet::LayerLearningReport(void)
|
|
|
|
|
{
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID)
|
|
|
|
|
return "";
|
|
|
|
|
int total = layers.Total();
|
|
|
|
|
if(ArraySize(m_prevLayerNorm) != total)
|
|
|
|
|
{
|
|
|
|
|
ArrayResize(m_prevLayerNorm, total);
|
|
|
|
|
ArrayInitialize(m_prevLayerNorm, -1.0);
|
|
|
|
|
}
|
|
|
|
|
//--- One slot per layer index, so At(l) lines up with the loop below without bookkeeping. Layers that
|
|
|
|
|
//--- own no weights just keep an empty array.
|
|
|
|
|
if(CheckPointer(m_prevLayerWeights) == POINTER_INVALID)
|
|
|
|
|
m_prevLayerWeights = new CArrayObj();
|
|
|
|
|
if(CheckPointer(m_prevLayerWeights) != POINTER_INVALID)
|
|
|
|
|
while(m_prevLayerWeights.Total() < total)
|
|
|
|
|
if(!m_prevLayerWeights.Add(new CArrayDouble()))
|
|
|
|
|
break;
|
|
|
|
|
string report = "";
|
|
|
|
|
for(int l = 0; l < total; l++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *layer = (CLayer*)layers.At(l);
|
|
|
|
|
if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0)
|
|
|
|
|
continue;
|
|
|
|
|
CObject *obj = layer.At(0);
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
double w[];
|
|
|
|
|
int got = 0;
|
|
|
|
|
string tag = "?";
|
|
|
|
|
switch(obj.Type())
|
|
|
|
|
{
|
|
|
|
|
case defNeuronBaseOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj;
|
|
|
|
|
got = n.getWeights(w);
|
|
|
|
|
tag = "dense";
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronConvOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
|
|
|
|
|
got = c.getWeightsConv(w);
|
|
|
|
|
tag = "conv";
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronLSTMOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
|
|
|
|
|
got = ls.getWeightsLSTM(w);
|
|
|
|
|
tag = "lstm";
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronBatchNormOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
|
|
|
|
|
got = bn.getWeightsBN(w);
|
|
|
|
|
tag = "bn";
|
2026-08-02 08:12:47 -04:00
|
|
|
//--- getWeightsBN packs FOUR different kinds of number into one array: the outgoing dense
|
|
|
|
|
//--- matrix, the learned gamma/beta, the running mean/variance, and the Adam moment
|
|
|
|
|
//--- buffers. A single norm over all of them cannot say which one is moving - and on
|
|
|
|
|
//--- 2026-08-02 that ambiguity was the difference between "the weights are diverging"
|
|
|
|
|
//--- (an optimizer problem) and "the running variance is tracking activations that grew"
|
|
|
|
|
//--- (a scaling problem), which need opposite fixes. PAI reached bn3:465,684 with no way
|
|
|
|
|
//--- to tell them apart. So break the block down: the sub-norms cost one pass and turn a
|
|
|
|
|
//--- guess into a reading.
|
|
|
|
|
int oCount = bn.BatchOptionsTotal();
|
|
|
|
|
int wCount = got - oCount;
|
|
|
|
|
if(oCount > 0 && wCount >= 0)
|
|
|
|
|
{
|
|
|
|
|
double nW = 0.0, nGamma = 0.0, nBeta = 0.0, nMean = 0.0, nVar = 0.0, nAdam = 0.0;
|
|
|
|
|
for(int i = 0; i < wCount; i++)
|
|
|
|
|
nW += w[i] * w[i];
|
|
|
|
|
for(int o = 0; o < oCount; o++)
|
|
|
|
|
{
|
|
|
|
|
double v = w[wCount + o];
|
|
|
|
|
switch(o % BN_OPT_STRIDE)
|
|
|
|
|
{
|
|
|
|
|
case BN_OPT_MEAN: nMean += v * v; break;
|
|
|
|
|
case BN_OPT_VAR: nVar += v * v; break;
|
|
|
|
|
case BN_OPT_GAMMA: nGamma += v * v; break;
|
|
|
|
|
case BN_OPT_BETA: nBeta += v * v; break;
|
|
|
|
|
case BN_OPT_MG:
|
|
|
|
|
case BN_OPT_MB:
|
|
|
|
|
case BN_OPT_VG:
|
|
|
|
|
case BN_OPT_VB: nAdam += v * v; break;
|
|
|
|
|
default: break; // BN_OPT_NX is a forward-pass scratch value
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
report += StringFormat(" bn%d[W %.3g|g %.3g|b %.3g|mean %.3g|var %.3g|adam %.3g]",
|
|
|
|
|
l, MathSqrt(nW), MathSqrt(nGamma), MathSqrt(nBeta),
|
|
|
|
|
MathSqrt(nMean), MathSqrt(nVar), MathSqrt(nAdam));
|
|
|
|
|
}
|
2026-08-01 11:27:28 -04:00
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronPoolOCL:
|
|
|
|
|
//--- No parameters of its own. Named anyway so the report shows the real layer order.
|
|
|
|
|
report += " pool" + IntegerToString(l) + ":-";
|
|
|
|
|
continue;
|
|
|
|
|
default:
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
if(got <= 0)
|
|
|
|
|
{
|
|
|
|
|
report += " " + tag + IntegerToString(l) + ":NOWEIGHTS";
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
double sum = 0.0;
|
|
|
|
|
for(int i = 0; i < got; i++)
|
|
|
|
|
sum += w[i] * w[i];
|
|
|
|
|
double norm = MathSqrt(sum);
|
|
|
|
|
double prev = m_prevLayerNorm[l];
|
|
|
|
|
m_prevLayerNorm[l] = norm;
|
|
|
|
|
//--- |dW|: norm of the elementwise change since the last report. Only meaningful when the previous
|
|
|
|
|
//--- vector has the same length (a rebuilt topology invalidates it), hence the size check.
|
|
|
|
|
double stepRel = -1.0;
|
|
|
|
|
CArrayDouble *prevW = (CheckPointer(m_prevLayerWeights) != POINTER_INVALID && l < m_prevLayerWeights.Total()
|
|
|
|
|
? (CArrayDouble*)m_prevLayerWeights.At(l) : NULL);
|
|
|
|
|
if(CheckPointer(prevW) != POINTER_INVALID)
|
|
|
|
|
{
|
|
|
|
|
if(prevW.Total() == got && prev > 0.0)
|
|
|
|
|
{
|
|
|
|
|
double d2 = 0.0;
|
|
|
|
|
for(int i = 0; i < got; i++)
|
|
|
|
|
{
|
|
|
|
|
double d = w[i] - prevW.At(i);
|
|
|
|
|
d2 += d * d;
|
|
|
|
|
}
|
|
|
|
|
stepRel = MathSqrt(d2) / prev;
|
|
|
|
|
}
|
|
|
|
|
prevW.Clear();
|
|
|
|
|
for(int i = 0; i < got; i++)
|
|
|
|
|
prevW.Add(w[i]);
|
|
|
|
|
}
|
|
|
|
|
if(prev < 0.0)
|
|
|
|
|
{
|
|
|
|
|
report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) + "(init)";
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
//--- Relative, so a wide layer and a narrow one are comparable at a glance. Printed as
|
|
|
|
|
//--- norm(d|W| / |dW|): the FIRST is how much the length changed, the SECOND how far the vector
|
|
|
|
|
//--- actually moved. Decay-only shows the two roughly EQUAL with the norm falling; a learning
|
|
|
|
|
//--- layer shows the second clearly larger. See m_prevLayerWeights' declaration comment.
|
|
|
|
|
double rel = (prev > 0.0 ? MathAbs(norm - prev) / prev : 0.0);
|
|
|
|
|
report += " " + tag + IntegerToString(l) + ":" + DoubleToString(norm, 3) +
|
|
|
|
|
"(" + DoubleToString(100.0 * rel, 3) + "%/" +
|
|
|
|
|
(stepRel < 0.0 ? "n/a" : DoubleToString(100.0 * stepRel, 3) + "%") + ")";
|
|
|
|
|
}
|
|
|
|
|
return report;
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| Snapshot every neuron's weights into host memory (see the header |
|
|
|
|
|
//| declaration). Layer-major order; one CArrayDouble per neuron. Used|
|
|
|
|
|
//| as the mid-run best-era checkpoint - restored by RestoreWeights().|
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
bool CNet::CaptureWeights(void)
|
|
|
|
|
{
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
if(CheckPointer(m_weightSnapshot) == POINTER_INVALID)
|
|
|
|
|
{
|
|
|
|
|
m_weightSnapshot = new CArrayObj();
|
|
|
|
|
if(CheckPointer(m_weightSnapshot) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
m_weightSnapshot.Clear(); // FreeMode deletes the previous snapshot's per-neuron arrays
|
|
|
|
|
m_haveWeightSnapshot = false;
|
|
|
|
|
for(int l = 0; l < layers.Total(); l++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *layer = (CLayer*)layers.At(l);
|
|
|
|
|
if(CheckPointer(layer) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
for(int n = 0; n < layer.Total(); n++)
|
|
|
|
|
{
|
|
|
|
|
CObject *obj = layer.At(n);
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
double w[];
|
|
|
|
|
int got = 0;
|
|
|
|
|
switch(obj.Type())
|
|
|
|
|
{
|
|
|
|
|
case defNeuronBaseOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj;
|
|
|
|
|
got = neuron.getWeights(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronBatchNormOCL:
|
|
|
|
|
{
|
|
|
|
|
//--- Dense matrix + gamma/beta/statistics as one array - see getWeightsBN. Without this
|
|
|
|
|
//--- the plateau ladder would restore the weights around this layer while leaving its
|
|
|
|
|
//--- own parameters at whatever the diverged era left behind.
|
|
|
|
|
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
|
|
|
|
|
got = bn.getWeightsBN(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronConvOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
|
|
|
|
|
got = c.getWeightsConv(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronLSTMOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
|
|
|
|
|
got = ls.getWeightsLSTM(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuron:
|
|
|
|
|
case defNeuronConv:
|
|
|
|
|
case defNeuronPool:
|
|
|
|
|
case defNeuronLSTM:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBase *neuron = (CNeuronBase*)obj;
|
|
|
|
|
CArrayCon *conns = neuron.getConnections();
|
|
|
|
|
if(CheckPointer(conns) != POINTER_INVALID)
|
|
|
|
|
{
|
|
|
|
|
got = conns.Total();
|
|
|
|
|
ArrayResize(w, got);
|
|
|
|
|
for(int i = 0; i < got; i++)
|
|
|
|
|
{
|
|
|
|
|
CConnection *con = conns.At(i);
|
|
|
|
|
w[i] = (CheckPointer(con) != POINTER_INVALID) ? con.weight : 0.0;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
default:
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
CArrayDouble *snap = new CArrayDouble();
|
|
|
|
|
if(CheckPointer(snap) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
if(got > 0)
|
|
|
|
|
snap.AssignArray(w);
|
|
|
|
|
if(!m_weightSnapshot.Add(snap))
|
|
|
|
|
{
|
|
|
|
|
delete snap;
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
m_haveWeightSnapshot = true;
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| Write the CaptureWeights() snapshot back into the live neurons IN |
|
|
|
|
|
//| PLACE (setWeights - no neuron re-creation, no new device tensors).|
|
|
|
|
|
//| Aborts (returning false, live weights untouched past that point) |
|
|
|
|
|
//| only on a neuron-count mismatch, which never happens within a run |
|
|
|
|
|
//| (fixed topology). See the header declaration for the full why. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
bool CNet::RestoreWeights(void)
|
|
|
|
|
{
|
|
|
|
|
if(!m_haveWeightSnapshot || CheckPointer(m_weightSnapshot) == POINTER_INVALID || CheckPointer(layers) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
int idx = 0;
|
|
|
|
|
for(int l = 0; l < layers.Total(); l++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *layer = (CLayer*)layers.At(l);
|
|
|
|
|
if(CheckPointer(layer) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
for(int n = 0; n < layer.Total(); n++)
|
|
|
|
|
{
|
|
|
|
|
if(idx >= m_weightSnapshot.Total())
|
|
|
|
|
return false; // topology/count mismatch - stop rather than mis-map weights
|
|
|
|
|
CObject *obj = layer.At(n);
|
|
|
|
|
CArrayDouble *snap = m_weightSnapshot.At(idx);
|
|
|
|
|
idx++;
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID || CheckPointer(snap) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
int cnt = snap.Total();
|
|
|
|
|
if(cnt <= 0)
|
|
|
|
|
continue; // no weights captured for this neuron (e.g. output layer) - nothing to restore
|
|
|
|
|
double w[];
|
|
|
|
|
ArrayResize(w, cnt);
|
|
|
|
|
for(int i = 0; i < cnt; i++)
|
|
|
|
|
w[i] = snap.At(i);
|
|
|
|
|
switch(obj.Type())
|
|
|
|
|
{
|
|
|
|
|
case defNeuronBaseOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBaseOCL *neuron = (CNeuronBaseOCL*)obj;
|
|
|
|
|
neuron.setWeights(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronBatchNormOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
|
|
|
|
|
bn.setWeightsBN(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronConvOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronConvOCL *c = (CNeuronConvOCL*)obj;
|
|
|
|
|
c.setWeightsConv(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuronLSTMOCL:
|
|
|
|
|
{
|
|
|
|
|
CNeuronLSTMOCL *ls = (CNeuronLSTMOCL*)obj;
|
|
|
|
|
ls.setWeightsLSTM(w);
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
case defNeuron:
|
|
|
|
|
case defNeuronConv:
|
|
|
|
|
case defNeuronPool:
|
|
|
|
|
case defNeuronLSTM:
|
|
|
|
|
{
|
|
|
|
|
CNeuronBase *neuron = (CNeuronBase*)obj;
|
|
|
|
|
CArrayCon *conns = neuron.getConnections();
|
|
|
|
|
if(CheckPointer(conns) == POINTER_INVALID)
|
|
|
|
|
break;
|
|
|
|
|
int count = MathMin(cnt, conns.Total());
|
|
|
|
|
for(int i = 0; i < count; i++)
|
|
|
|
|
{
|
|
|
|
|
CConnection *con = conns.At(i);
|
|
|
|
|
if(CheckPointer(con) != POINTER_INVALID)
|
|
|
|
|
con.weight = w[i];
|
|
|
|
|
}
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
default:
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return (idx == m_weightSnapshot.Total());
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| See this method's declaration comment for the cold-start bias |
|
|
|
|
|
//| rationale. The weight block that produces the output layer's |
|
|
|
|
|
//| values is stored on the layer BEFORE it (see CNeuronBaseOCL:: |
|
|
|
|
|
//| feedForward(CNeuronBaseOCL*) - matrix_w comes from the SOURCE |
|
|
|
|
|
//| neuron, laid out as (sourceNeurons+1) values per destination |
|
|
|
|
|
//| neuron, the last of which is that neuron's bias term), so this |
|
|
|
|
|
//| reaches one layer back from the output layer to edit it. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
bool CNet::SeedOutputLayerBias(const double &biasValues[])
|
|
|
|
|
{
|
|
|
|
|
int outputs = ArraySize(biasValues);
|
|
|
|
|
if(outputs <= 0 || CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2)
|
|
|
|
|
return false;
|
|
|
|
|
CLayer *sourceLayer = layers.At(layers.Total() - 2);
|
|
|
|
|
if(CheckPointer(sourceLayer) == POINTER_INVALID || sourceLayer.Total() <= 0)
|
|
|
|
|
return false;
|
|
|
|
|
CObject *sourceObj = sourceLayer.At(0);
|
|
|
|
|
//--- Batch norm is accepted here as well as a plain dense layer. It is a CNeuronBaseOCL subclass
|
|
|
|
|
//--- with the identical (inputs+1)*outputs weight layout - it just happens to be the layer that
|
|
|
|
|
//--- carries the head's weight matrix once normalization is enabled (see AI\NeuronBatchNorm.mqh and
|
|
|
|
|
//--- BuildFreshTopology's batch-norm insertion). Without this the exact-type check silently failed
|
|
|
|
|
//--- and the cold-start output-bias seed stopped being applied for every batch-norm topology.
|
|
|
|
|
if(CheckPointer(sourceObj) == POINTER_INVALID ||
|
|
|
|
|
(sourceObj.Type() != defNeuronBaseOCL && sourceObj.Type() != defNeuronBatchNormOCL))
|
|
|
|
|
return false;
|
|
|
|
|
CNeuronBaseOCL *sourceNeuron = (CNeuronBaseOCL*)sourceObj;
|
|
|
|
|
if(CheckPointer(sourceNeuron) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
int inputs = sourceNeuron.Neurons();
|
|
|
|
|
double weights[];
|
|
|
|
|
int count = sourceNeuron.getWeights(weights);
|
|
|
|
|
if(count != (inputs + 1) * outputs)
|
|
|
|
|
return false; // layout doesn't match the assumed dense (source+1)*outputs block - don't guess
|
|
|
|
|
for(int i = 0; i < outputs; i++)
|
|
|
|
|
weights[(inputs + 1) * i + inputs] = biasValues[i];
|
|
|
|
|
return sourceNeuron.setWeights(weights);
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
void CNet::SetBatchNormFrozen(bool frozen)
|
|
|
|
|
{
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID)
|
|
|
|
|
return;
|
|
|
|
|
for(int l = 0; l < layers.Total(); l++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *layer = (CLayer*)layers.At(l);
|
|
|
|
|
if(CheckPointer(layer) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
for(int n = 0; n < layer.Total(); n++)
|
|
|
|
|
{
|
|
|
|
|
CObject *obj = layer.At(n);
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID || obj.Type() != defNeuronBatchNormOCL)
|
|
|
|
|
continue;
|
|
|
|
|
CNeuronBatchNormOCL *bn = (CNeuronBatchNormOCL*)obj;
|
|
|
|
|
bn.SetStatsFrozen(frozen);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| Window of the FIRST convolutional layer in the loaded net, or 0 |
|
|
|
|
|
//| when there is none. |
|
|
|
|
|
//| |
|
|
|
|
|
//| Same problem as EnforceOutputActivation below, different field: a |
|
|
|
|
|
//| .nnw stores the window/step each conv layer was BUILT with, so a |
|
|
|
|
|
//| model saved before the receptive field changed keeps the old |
|
|
|
|
|
//| shape forever and goes on training under an architecture the code |
|
|
|
|
|
//| no longer specifies. Unlike the activation this CANNOT be |
|
|
|
|
|
//| repaired in place - the weight block is (window+1)*window_out and |
|
|
|
|
|
//| a different window is a different tensor - so the caller's only |
|
|
|
|
|
//| correct response is to retrain. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
uint CNet::FirstConvWindow(void)
|
|
|
|
|
{
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID)
|
|
|
|
|
return 0;
|
|
|
|
|
for(int i = 0; i < layers.Total(); i++)
|
|
|
|
|
{
|
|
|
|
|
CLayer *layer = layers.At(i);
|
|
|
|
|
if(CheckPointer(layer) == POINTER_INVALID || layer.Total() <= 0)
|
|
|
|
|
continue;
|
|
|
|
|
CObject *obj = layer.At(0);
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID)
|
|
|
|
|
continue;
|
|
|
|
|
if(obj.Type() == defNeuronConvOCL)
|
|
|
|
|
{
|
|
|
|
|
CNeuronConvOCL *conv = (CNeuronConvOCL*)obj;
|
|
|
|
|
return conv.Window();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
//| See the declaration comment for why a loaded model's activation |
|
|
|
|
|
//| cannot be trusted and must be re-asserted from the topology spec. |
|
|
|
|
|
//+------------------------------------------------------------------+
|
|
|
|
|
bool CNet::EnforceOutputActivation(ENUM_ACTIVATION intended, ENUM_ACTIVATION &previous)
|
|
|
|
|
{
|
|
|
|
|
previous = intended;
|
|
|
|
|
if(CheckPointer(layers) == POINTER_INVALID || layers.Total() < 2)
|
|
|
|
|
return false;
|
|
|
|
|
CLayer *outputLayer = layers.At(layers.Total() - 1);
|
|
|
|
|
if(CheckPointer(outputLayer) == POINTER_INVALID || outputLayer.Total() <= 0)
|
|
|
|
|
return false;
|
|
|
|
|
CObject *obj = outputLayer.At(0);
|
|
|
|
|
if(CheckPointer(obj) == POINTER_INVALID)
|
|
|
|
|
return false;
|
|
|
|
|
//--- Both neuron models expose the same two accessors (CNeuronBase gained Activation() for exactly
|
|
|
|
|
//--- this), so one branch per family is enough - and anything else (conv/pool/LSTM can never be an
|
|
|
|
|
//--- output layer in this project's topologies) is left untouched rather than force-cast.
|
|
|
|
|
int t = obj.Type();
|
|
|
|
|
if(t == defNeuronBaseOCL || t == defNeuronConvOCL || t == defNeuronPoolOCL || t == defNeuronLSTMOCL ||
|
|
|
|
|
t == defNeuronBatchNormOCL)
|
|
|
|
|
{
|
|
|
|
|
CNeuronBaseOCL *n = (CNeuronBaseOCL*)obj;
|
|
|
|
|
previous = n.Activation();
|
|
|
|
|
if(previous == intended)
|
|
|
|
|
return false;
|
|
|
|
|
n.SetActivationFunction(intended);
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
if(t == defNeuron || t == defNeuronConv || t == defNeuronPool || t == defNeuronLSTM)
|
|
|
|
|
{
|
|
|
|
|
//--- The scalar CPU model stores activation per NEURON, not per layer, so every neuron in the
|
|
|
|
|
//--- output layer has to be corrected - not just index 0.
|
|
|
|
|
bool repaired = false;
|
|
|
|
|
for(int i = 0; i < outputLayer.Total(); i++)
|
|
|
|
|
{
|
|
|
|
|
CObject *cell = outputLayer.At(i);
|
|
|
|
|
if(CheckPointer(cell) == POINTER_INVALID || cell.Type() != defNeuron)
|
|
|
|
|
continue;
|
|
|
|
|
CNeuronBase *n = (CNeuronBase*)cell;
|
|
|
|
|
if(n.Activation() == intended)
|
|
|
|
|
continue;
|
|
|
|
|
if(!repaired)
|
|
|
|
|
previous = n.Activation();
|
|
|
|
|
n.SetActivationFunction(intended);
|
|
|
|
|
repaired = true;
|
|
|
|
|
}
|
|
|
|
|
return repaired;
|
|
|
|
|
}
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
#endif
|