544 lines
26 KiB
MQL5
544 lines
26 KiB
MQL5
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| NetBuild.mqh |
|
||
|
|
//| |
|
||
|
|
//| CNet lifecycle: statics, topology construction from |
|
||
|
|
//| CLayerDescription, OpenCL/DirectML init, destructor. |
|
||
|
|
//| |
|
||
|
|
//| Included from AI\Network.mqh AFTER every class declaration - |
|
||
|
|
//| bodies only, no declarations. Relocation is behaviour-neutral by |
|
||
|
|
//| construction: nothing here is reachable until Network.mqh ends. |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
#ifndef WARRIOR_AI_IMPL_NETBUILD_MQH
|
||
|
|
#define WARRIOR_AI_IMPL_NETBUILD_MQH
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
double CNet::recentAverageSmoothingFactor = 10000.0; // Number of training samples to average over
|
||
|
|
bool CNet::s_openclUnavailable = false;
|
||
|
|
bool CNet::s_computeTierLogged = false;
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| Element width that a conv or pool stage actually slides over. |
|
||
|
|
//| |
|
||
|
|
//| Reads the REAL output width of the layer already built below it, |
|
||
|
|
//| rather than the running position cursor. This matters the moment |
|
||
|
|
//| two window stages are stacked: a conv's own output is |
|
||
|
|
//| units_count * window_out (CNeuronConvOCL::Init), but the cursor |
|
||
|
|
//| tracks sliding POSITIONS only, so it under-reports a conv's width |
|
||
|
|
//| by a factor of window_out. A pool sized off that cursor would |
|
||
|
|
//| reduce the wrong number of elements and silently build a layer of |
|
||
|
|
//| the wrong shape - the failure would look like bad accuracy, not a |
|
||
|
|
//| crash. Same technique, and the same reason, as the batch-norm |
|
||
|
|
//| branch in the constructor below. |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
int ConvChainInputWidth(CArrayLayer *builtLayers, CLayerDescription *prev, CLayerDescription *desc, int cursor)
|
||
|
|
{
|
||
|
|
if(CheckPointer(builtLayers) != POINTER_INVALID && builtLayers.Total() > 0)
|
||
|
|
{
|
||
|
|
CLayer *below = builtLayers.At(builtLayers.Total() - 1);
|
||
|
|
if(CheckPointer(below) != POINTER_INVALID && below.Total() > 0)
|
||
|
|
{
|
||
|
|
CNeuronBaseOCL *belowNeuron = below.At(0);
|
||
|
|
if(CheckPointer(belowNeuron) != POINTER_INVALID && belowNeuron.Neurons() > 0)
|
||
|
|
return (int)belowNeuron.Neurons();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
//--- Nothing built yet (a window stage at index 0) - fall back to the description arithmetic.
|
||
|
|
if(CheckPointer(prev) == POINTER_INVALID)
|
||
|
|
return (CheckPointer(desc) == POINTER_INVALID ? cursor : (int)desc.count);
|
||
|
|
if(prev.type == defNeuron || prev.type == defNeuronBaseOCL)
|
||
|
|
return (int)prev.count;
|
||
|
|
return cursor;
|
||
|
|
}
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
CNet::CNet(CArrayObj *Description)
|
||
|
|
{
|
||
|
|
//--- Before the early returns below: CNet(NULL) is a legitimate construction (see
|
||
|
|
//--- InitNeuralNetwork) and must still leave the adjustment cleanly disabled.
|
||
|
|
bLogitAdjust = false;
|
||
|
|
ArrayInitialize(dLogitAdjust, 0.0);
|
||
|
|
if(CheckPointer(Description) == POINTER_INVALID)
|
||
|
|
return;
|
||
|
|
//---
|
||
|
|
int total = Description.Total();
|
||
|
|
if(total <= 0)
|
||
|
|
return;
|
||
|
|
//---
|
||
|
|
layers = new CArrayLayer();
|
||
|
|
if(CheckPointer(layers) == POINTER_INVALID)
|
||
|
|
return;
|
||
|
|
//---
|
||
|
|
CLayer *temp;
|
||
|
|
CLayerDescription *desc = NULL, *next = NULL, *prev = NULL;
|
||
|
|
CNeuronBase *neuron = NULL;
|
||
|
|
CNeuronPool *neuron_p = NULL;
|
||
|
|
int output_count = 0;
|
||
|
|
int temp_count = 0;
|
||
|
|
//---
|
||
|
|
next = Description.At(1);
|
||
|
|
if(CheckPointer(next) != POINTER_INVALID &&
|
||
|
|
(next.type == defNeuron || next.type == defNeuronBaseOCL || next.type == defNeuronConv || next.type == defNeuronConvOCL ||
|
||
|
|
next.type == defNeuronLSTM || next.type == defNeuronBatchNorm || next.type == defNeuronBatchNormOCL))
|
||
|
|
{
|
||
|
|
//--- OpenCL first, DirectML/D3D12 next, plain CPU as the final fallback
|
||
|
|
if(!InitOpenCL())
|
||
|
|
InitDirectML();
|
||
|
|
}
|
||
|
|
//--- Batch normalization exists only in the OCL neuron model (AI\NeuronBatchNorm.mqh); the legacy
|
||
|
|
//--- scalar CNeuron path below has no counterpart. Rather than silently build a DIFFERENT network
|
||
|
|
//--- than the topology asked for - which is precisely the class of bug that had models training
|
||
|
|
//--- against a stale output head for a day - refuse, loudly, and leave an empty net behind. Callers
|
||
|
|
//--- already treat a 0-layer net as a hard failure. Reaching here means OpenCL, DirectML AND the
|
||
|
|
//--- CPU DLL all failed to initialize, which this project does not support for training anyway.
|
||
|
|
if(CheckPointer(opencl) == POINTER_INVALID && CheckPointer(directml) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
for(int i = 0; i < total; i++)
|
||
|
|
{
|
||
|
|
CLayerDescription *probe = Description.At(i);
|
||
|
|
if(CheckPointer(probe) != POINTER_INVALID &&
|
||
|
|
(probe.type == defNeuronBatchNorm || probe.type == defNeuronBatchNormOCL))
|
||
|
|
{
|
||
|
|
Print(__FUNCTION__ + ": REFUSED - topology requests a batch-normalization layer but no compute"
|
||
|
|
" backend initialized (no OpenCL, no DirectML, no CPU DLL). Batch norm has no scalar-CPU"
|
||
|
|
" implementation; building without it would silently train a different architecture.");
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
//---
|
||
|
|
for(int i = 0; i < total; i++)
|
||
|
|
{
|
||
|
|
prev = desc;
|
||
|
|
desc = Description.At(i);
|
||
|
|
if((i + 1) < total)
|
||
|
|
{
|
||
|
|
next = Description.At(i + 1);
|
||
|
|
if(CheckPointer(next) == POINTER_INVALID)
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
else
|
||
|
|
next = NULL;
|
||
|
|
//--- How many outgoing weights this layer carries. The convention in this engine is that the
|
||
|
|
//--- weight matrix feeding layer L is stored ON layer L-1, so only a DENSE successor claims one:
|
||
|
|
//--- conv/pool/LSTM own their weights internally, and batch norm is elementwise and has none at
|
||
|
|
//--- all (just gamma/beta, which live in its own parameter block). Getting this wrong for the new
|
||
|
|
//--- type would allocate a full dense matrix that nothing ever reads or trains.
|
||
|
|
int outputs = (next == NULL || (next.type != defNeuron && next.type != defNeuronBaseOCL) ? 0 : next.count);
|
||
|
|
temp = new CLayer(outputs);
|
||
|
|
int neurons = (desc.count + (desc.type == defNeuron || desc.type == defNeuronBaseOCL ? 1 : 0));
|
||
|
|
if(CheckPointer(opencl) != POINTER_INVALID || CheckPointer(directml) != POINTER_INVALID)
|
||
|
|
{
|
||
|
|
CNeuronBaseOCL *neuron_ocl = NULL;
|
||
|
|
switch(desc.type)
|
||
|
|
{
|
||
|
|
case defNeuron:
|
||
|
|
case defNeuronBaseOCL:
|
||
|
|
neuron_ocl = new CNeuronBaseOCL();
|
||
|
|
if(CheckPointer(neuron_ocl) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
if(CheckPointer(opencl) != POINTER_INVALID
|
||
|
|
? !neuron_ocl.Init(outputs, 0, opencl, desc.count, desc.optimization)
|
||
|
|
: !neuron_ocl.Init(outputs, 0, directml, desc.count, desc.optimization))
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_ocl.SetActivationFunction(desc.activation);
|
||
|
|
if(!temp.Add(neuron_ocl))
|
||
|
|
{
|
||
|
|
delete neuron_ocl;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_ocl = NULL;
|
||
|
|
break;
|
||
|
|
case defNeuronBatchNorm:
|
||
|
|
case defNeuronBatchNormOCL:
|
||
|
|
{
|
||
|
|
CNeuronBatchNormOCL *neuron_bn = new CNeuronBatchNormOCL();
|
||
|
|
if(CheckPointer(neuron_bn) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
//--- Elementwise, so this layer is exactly as wide as the one below it. Take that width
|
||
|
|
//--- from the layer already built rather than from desc.count: a conv or pool stage's
|
||
|
|
//--- output size is derived HERE (the sliding-window arithmetic above), so the topology
|
||
|
|
//--- builder in ExpertSignalAIBase.mqh has no way to know it and cannot state it. Falls
|
||
|
|
//--- back to desc.count only for the impossible case of a batch-norm layer at index 0.
|
||
|
|
int bnUnits = desc.count;
|
||
|
|
if(layers.Total() > 0)
|
||
|
|
{
|
||
|
|
CLayer *below = layers.At(layers.Total() - 1);
|
||
|
|
if(CheckPointer(below) != POINTER_INVALID && below.Total() > 0)
|
||
|
|
{
|
||
|
|
CNeuronBaseOCL *belowNeuron = below.At(0);
|
||
|
|
if(CheckPointer(belowNeuron) != POINTER_INVALID && belowNeuron.Neurons() > 0)
|
||
|
|
bnUnits = belowNeuron.Neurons();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
bool bnInit = (CheckPointer(opencl) != POINTER_INVALID
|
||
|
|
? neuron_bn.Init(outputs, 0, opencl, bnUnits, desc.batch, desc.optimization)
|
||
|
|
: neuron_bn.Init(outputs, 0, directml, bnUnits, desc.batch, desc.optimization));
|
||
|
|
if(!bnInit)
|
||
|
|
{
|
||
|
|
delete neuron_bn;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
if(!temp.Add(neuron_bn))
|
||
|
|
{
|
||
|
|
delete neuron_bn;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_bn = NULL;
|
||
|
|
//--- Keep the running conv/pool sizing cursor pointing at this layer's real width, so a
|
||
|
|
//--- conv or pool stage placed ABOVE a batch-norm layer still sizes correctly.
|
||
|
|
output_count = bnUnits;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
case defNeuronConv:
|
||
|
|
case defNeuronConvOCL:
|
||
|
|
{
|
||
|
|
CNeuronConvOCL *neuron_conv = new CNeuronConvOCL();
|
||
|
|
if(CheckPointer(neuron_conv) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
//--- number of sliding positions - same formula the CPU CNeuronConv path uses.
|
||
|
|
//--- output_count keeps meaning POSITIONS (it is this layer's units_count); the width
|
||
|
|
//--- being slid over comes from the built layer below - see ConvChainInputWidth.
|
||
|
|
int convIn = ConvChainInputWidth(layers, prev, desc, output_count);
|
||
|
|
temp_count = (convIn - desc.window) % desc.step;
|
||
|
|
output_count = (convIn - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2);
|
||
|
|
bool convInit = (CheckPointer(opencl) != POINTER_INVALID
|
||
|
|
? neuron_conv.Init(outputs, 0, opencl, desc.window, desc.step, desc.count, output_count, desc.optimization)
|
||
|
|
: neuron_conv.Init(outputs, 0, directml, desc.window, desc.step, desc.count, output_count, desc.optimization));
|
||
|
|
if(!convInit)
|
||
|
|
{
|
||
|
|
delete neuron_conv;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_conv.SetActivationFunction(desc.activation);
|
||
|
|
if(!temp.Add(neuron_conv))
|
||
|
|
{
|
||
|
|
delete neuron_conv;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_conv = NULL;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
case defNeuronPool:
|
||
|
|
case defNeuronPoolOCL:
|
||
|
|
{
|
||
|
|
CNeuronPoolOCL *neuron_pool = new CNeuronPoolOCL();
|
||
|
|
if(CheckPointer(neuron_pool) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
//--- number of sliding positions - same formula the CPU CNeuronPool path uses.
|
||
|
|
//--- Critically this must slide over the conv's FULL output (positions * filters), which
|
||
|
|
//--- is what ConvChainInputWidth returns; the position cursor alone would be window_out
|
||
|
|
//--- times too small and pool the wrong element count entirely.
|
||
|
|
int poolIn = ConvChainInputWidth(layers, prev, desc, output_count);
|
||
|
|
temp_count = (poolIn - desc.window) % desc.step;
|
||
|
|
output_count = (poolIn - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2);
|
||
|
|
bool poolInit = (CheckPointer(opencl) != POINTER_INVALID
|
||
|
|
? neuron_pool.Init(outputs, 0, opencl, desc.window, desc.step, output_count, desc.optimization)
|
||
|
|
: neuron_pool.Init(outputs, 0, directml, desc.window, desc.step, output_count, desc.optimization));
|
||
|
|
if(!poolInit)
|
||
|
|
{
|
||
|
|
delete neuron_pool;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
if(!temp.Add(neuron_pool))
|
||
|
|
{
|
||
|
|
delete neuron_pool;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_pool = NULL;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
case defNeuronLSTM:
|
||
|
|
case defNeuronLSTMOCL:
|
||
|
|
{
|
||
|
|
CNeuronLSTMOCL *neuron_lstm = new CNeuronLSTMOCL();
|
||
|
|
if(CheckPointer(neuron_lstm) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
bool lstmInit = (CheckPointer(opencl) != POINTER_INVALID
|
||
|
|
? neuron_lstm.Init(outputs, 0, opencl, desc.count, desc.optimization)
|
||
|
|
: neuron_lstm.Init(outputs, 0, directml, desc.count, desc.optimization));
|
||
|
|
if(!lstmInit)
|
||
|
|
{
|
||
|
|
delete neuron_lstm;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
//--- desc.window carries the PER-TIMESTEP input width (see AddLstmStage) - the per-bar
|
||
|
|
//--- feature count reaching this layer. It used to be dead metadata; it is now what
|
||
|
|
//--- turns this into a recurrence over bars instead of one giant gated projection.
|
||
|
|
neuron_lstm.SetStepWidth(desc.window);
|
||
|
|
if(!temp.Add(neuron_lstm))
|
||
|
|
{
|
||
|
|
delete neuron_lstm;
|
||
|
|
delete temp;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron_lstm = NULL;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
default:
|
||
|
|
return;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
else
|
||
|
|
for(int n = 0; n < neurons; n++)
|
||
|
|
{
|
||
|
|
switch(desc.type)
|
||
|
|
{
|
||
|
|
case defNeuron:
|
||
|
|
neuron = new CNeuron();
|
||
|
|
if(CheckPointer(neuron) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
//--- He-scaled init, matching CNeuronBaseOCL::Init's rationale - fan-in is this
|
||
|
|
//--- layer's own neuron count (bias already included via the `neurons` count above).
|
||
|
|
neuron.Init(outputs, n, desc.optimization, MathSqrt(2.0 / (double)neurons));
|
||
|
|
neuron.SetActivationFunction(desc.activation);
|
||
|
|
break;
|
||
|
|
case defNeuronConv:
|
||
|
|
neuron_p = new CNeuronConv();
|
||
|
|
if(CheckPointer(neuron_p) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
if(CheckPointer(prev) != POINTER_INVALID)
|
||
|
|
{
|
||
|
|
if(prev.type == defNeuron)
|
||
|
|
{
|
||
|
|
temp_count = (int)((prev.count - desc.window) % desc.step);
|
||
|
|
output_count = (int)((prev.count - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2));
|
||
|
|
}
|
||
|
|
else
|
||
|
|
if(n == 0)
|
||
|
|
{
|
||
|
|
temp_count = (int)((output_count - desc.window) % desc.step);
|
||
|
|
output_count = (int)((output_count - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2));
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if(neuron_p.Init(outputs, n, desc.window, desc.step, output_count, desc.optimization))
|
||
|
|
neuron = neuron_p;
|
||
|
|
break;
|
||
|
|
case defNeuronPool:
|
||
|
|
neuron_p = new CNeuronPool();
|
||
|
|
if(CheckPointer(neuron_p) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
if(CheckPointer(prev) != POINTER_INVALID)
|
||
|
|
{
|
||
|
|
if(prev.type == defNeuron)
|
||
|
|
{
|
||
|
|
temp_count = (int)((prev.count - desc.window) % desc.step);
|
||
|
|
output_count = (int)((prev.count - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2));
|
||
|
|
}
|
||
|
|
else
|
||
|
|
if(n == 0)
|
||
|
|
{
|
||
|
|
temp_count = (int)((output_count - desc.window) % desc.step);
|
||
|
|
output_count = (int)((output_count - desc.window - temp_count) / desc.step + (temp_count == 0 ? 1 : 2));
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if(neuron_p.Init(outputs, n, desc.window, desc.step, output_count, desc.optimization))
|
||
|
|
neuron = neuron_p;
|
||
|
|
break;
|
||
|
|
case defNeuronLSTM:
|
||
|
|
neuron_p = new CNeuronLSTM();
|
||
|
|
if(CheckPointer(neuron_p) == POINTER_INVALID)
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
output_count = (next != NULL ? next.window : desc.step);
|
||
|
|
if(neuron_p.Init(outputs, n, desc.window, 1, output_count, desc.optimization))
|
||
|
|
neuron = neuron_p;
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
if(!temp.Add(neuron))
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
neuron = NULL;
|
||
|
|
}
|
||
|
|
if(!layers.Add(temp))
|
||
|
|
{
|
||
|
|
delete temp;
|
||
|
|
delete layers;
|
||
|
|
return;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
//---
|
||
|
|
}
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| Tries to initialize OpenCL; on any failure (no GPU, driver |
|
||
|
|
//| missing, kernel build error) frees it and leaves opencl==NULL |
|
||
|
|
//| so the rest of CNet transparently runs its CPU code path. |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
bool CNet::InitOpenCL(void)
|
||
|
|
{
|
||
|
|
//--- pure-MQL5 inference: deliberately refuse a backend so the OCL neurons compute host-side.
|
||
|
|
if(m_cpuInference)
|
||
|
|
return false;
|
||
|
|
if(CheckPointer(opencl) != POINTER_INVALID)
|
||
|
|
return true;
|
||
|
|
//--- already established (by an earlier CNet in this process) that this host has no OpenCL - skip the
|
||
|
|
//--- probe rather than re-run a known failure and reprint the stdlib's banner. See s_openclUnavailable.
|
||
|
|
if(s_openclUnavailable)
|
||
|
|
return false;
|
||
|
|
//---
|
||
|
|
opencl = new COpenCLMy();
|
||
|
|
if(CheckPointer(opencl) == POINTER_INVALID || !opencl.Initialize(cl_program, true))
|
||
|
|
{
|
||
|
|
if(CheckPointer(opencl) != POINTER_INVALID)
|
||
|
|
delete opencl;
|
||
|
|
opencl = NULL;
|
||
|
|
s_openclUnavailable = true;
|
||
|
|
PrintFormat("%s: OpenCL unavailable, falling back to CPU", __FUNCTION__);
|
||
|
|
return false;
|
||
|
|
}
|
||
|
|
//--- create kernels
|
||
|
|
opencl.SetKernelsCount(22);
|
||
|
|
opencl.KernelCreate(def_k_FeedForward, "FeedForward");
|
||
|
|
opencl.KernelCreate(def_k_CaclOutputGradient, "CaclOutputGradient");
|
||
|
|
opencl.KernelCreate(def_k_CaclHiddenGradient, "CaclHiddenGradient");
|
||
|
|
opencl.KernelCreate(def_k_UpdateWeightsMomentum, "UpdateWeightsMomentum");
|
||
|
|
opencl.KernelCreate(def_k_UpdateWeightsAdam, "UpdateWeightsAdam");
|
||
|
|
opencl.KernelCreate(def_k_FeedForwardConv, "FeedForwardConv");
|
||
|
|
opencl.KernelCreate(def_k_CalcHiddenGradientConv, "CalcHiddenGradientConv");
|
||
|
|
opencl.KernelCreate(def_k_UpdateWeightsConvMomentum, "UpdateWeightsConvMomentum");
|
||
|
|
opencl.KernelCreate(def_k_UpdateWeightsConvAdam, "UpdateWeightsConvAdam");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_Gates, "LSTM_Gates");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_State, "LSTM_State");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_GateGradient, "LSTM_GateGradient");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_WeightsGradient, "LSTM_WeightsGradient");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_InputsGradient, "LSTM_InputsGradient");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_UpdateWeightsAdam, "LSTM_UpdateWeightsAdam");
|
||
|
|
opencl.KernelCreate(def_k_LSTM_UpdateWeightsMomentum, "LSTM_UpdateWeightsMomentum");
|
||
|
|
opencl.KernelCreate(def_k_FeedForwardProof, "FeedForwardProof");
|
||
|
|
opencl.KernelCreate(def_k_CalcInputGradientProof, "CalcInputGradientProof");
|
||
|
|
//--- Return values CHECKED here, unlike the calls above. A kernel that fails to build otherwise
|
||
|
|
//--- surfaces only as an Execute() failure deep inside training on a customer's machine, and these
|
||
|
|
//--- four are the newest and least-exercised code in the program. Reported, not fatal: a device
|
||
|
|
//--- without them can still run every non-recurrent topology.
|
||
|
|
bool seqOk = opencl.KernelCreate(def_k_LSTM_SeqStepForward, "LSTM_SeqStepForward");
|
||
|
|
seqOk = opencl.KernelCreate(def_k_LSTM_SeqStepGateGrad, "LSTM_SeqStepGateGrad") && seqOk;
|
||
|
|
seqOk = opencl.KernelCreate(def_k_LSTM_SeqStepWeightGrad, "LSTM_SeqStepWeightGrad") && seqOk;
|
||
|
|
seqOk = opencl.KernelCreate(def_k_LSTM_SeqStepInputGrad, "LSTM_SeqStepInputGrad") && seqOk;
|
||
|
|
if(!seqOk)
|
||
|
|
Print("CNet::InitOpenCL: WARNING - the sequence-LSTM kernels failed to build on this OpenCL device. LSTM and HYBRID will not train here; MLP and CONV are unaffected. Run those topologies on the CPU/DirectML tier instead.");
|
||
|
|
return true;
|
||
|
|
}
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
//| Second-tier GPU fallback: only tried when OpenCL init failed. |
|
||
|
|
//| Requires DirectML\WarriorDML.dll in the terminal's Libraries |
|
||
|
|
//| folder (build it with DirectML\build.bat); on any failure frees |
|
||
|
|
//| itself and leaves directml==NULL so CNet falls through to CPU. |
|
||
|
|
//+------------------------------------------------------------------+
|
||
|
|
bool CNet::InitDirectML(void)
|
||
|
|
{
|
||
|
|
//--- pure-MQL5 inference: deliberately refuse a backend so the OCL neurons compute host-side.
|
||
|
|
if(m_cpuInference)
|
||
|
|
return false;
|
||
|
|
//--- Tester/optimization/forward runs must not touch WarriorDML/WarriorCPU DLL imports.
|
||
|
|
//--- This avoids agent-side file-lock/synchronization failures on rapid stop/restart cycles.
|
||
|
|
if(MQLInfoInteger(MQL_TESTER) || MQLInfoInteger(MQL_OPTIMIZATION) || MQLInfoInteger(MQL_FORWARD))
|
||
|
|
return false;
|
||
|
|
if(CheckPointer(directml) != POINTER_INVALID)
|
||
|
|
return true;
|
||
|
|
//---
|
||
|
|
directml = new CDirectMLMy();
|
||
|
|
if(CheckPointer(directml) == POINTER_INVALID)
|
||
|
|
return false;
|
||
|
|
directml.SetCpuLoadPercent(EffectiveCpuLoadPercent());
|
||
|
|
if(!directml.Initialize())
|
||
|
|
{
|
||
|
|
int err = directml.LastError();
|
||
|
|
delete directml;
|
||
|
|
directml = NULL;
|
||
|
|
string reason;
|
||
|
|
switch(err)
|
||
|
|
{
|
||
|
|
case 1: reason = "CreateDXGIFactory1 failed"; break;
|
||
|
|
case 2: reason = "no DX12 hardware adapter found (feature level 11_0)"; break;
|
||
|
|
case 3: reason = "compute command queue creation failed"; break;
|
||
|
|
case 4: reason = "command allocator creation failed"; break;
|
||
|
|
case 5: reason = "command list creation failed"; break;
|
||
|
|
case 6: reason = "fence creation failed"; break;
|
||
|
|
case 7: reason = "fence event creation failed"; break;
|
||
|
|
case 8: reason = "HLSL kernel compile/PSO creation failed"; break;
|
||
|
|
default: reason = "neither WarriorDML.dll nor WarriorCPU.dll loaded (check Libraries folder / \"Allow DLL imports\")";
|
||
|
|
}
|
||
|
|
PrintFormat("%s: DirectML/D3D12 and CPU DLL both unavailable (%s), falling back to slow per-object CPU path", __FUNCTION__, reason);
|
||
|
|
return false;
|
||
|
|
}
|
||
|
|
//--- Announce the tier ONCE per process: it describes the host, not this particular net, and a run builds
|
||
|
|
//--- several nets (main + EMA shadow + sim/self-check clones). See s_computeTierLogged.
|
||
|
|
if(!s_computeTierLogged)
|
||
|
|
{
|
||
|
|
s_computeTierLogged = true;
|
||
|
|
if(directml.Tier() == COMPUTE_TIER_CPU)
|
||
|
|
{
|
||
|
|
//--- Report the SPLIT, not just the result. When several charts train at once this is the
|
||
|
|
//--- number that explains their speed, and it is the one that was silently wrong before.
|
||
|
|
int share = EffectiveCpuLoadPercent();
|
||
|
|
PrintFormat("%s: DirectML/D3D12 unavailable, using multithreaded CPU DLL fallback (%d threads per network, target %d; %d%% of %d detected cores - fixed, independent of how many charts run)",
|
||
|
|
__FUNCTION__, directml.CpuThreadsUsed(), CPU_THREADS_PER_NETWORK, share, (int)TerminalInfoInteger(TERMINAL_CPU_CORES));
|
||
|
|
}
|
||
|
|
else
|
||
|
|
PrintFormat("%s: DirectML/D3D12 GPU tier active", __FUNCTION__);
|
||
|
|
}
|
||
|
|
return true;
|
||
|
|
}
|
||
|
|
CNet::~CNet(void)
|
||
|
|
{
|
||
|
|
if(CheckPointer(layers) != POINTER_INVALID)
|
||
|
|
delete layers;
|
||
|
|
if(CheckPointer(m_weightSnapshot) != POINTER_INVALID)
|
||
|
|
delete m_weightSnapshot; // CArrayObj (FreeMode) deletes its per-neuron CArrayDouble elements
|
||
|
|
if(CheckPointer(m_prevLayerWeights) != POINTER_INVALID)
|
||
|
|
delete m_prevLayerWeights; // same - FreeMode owns the per-layer CArrayDouble elements
|
||
|
|
if(CheckPointer(opencl) != POINTER_INVALID)
|
||
|
|
{
|
||
|
|
opencl.Shutdown();
|
||
|
|
delete opencl;
|
||
|
|
}
|
||
|
|
if(CheckPointer(directml) != POINTER_INVALID)
|
||
|
|
delete directml;
|
||
|
|
}
|
||
|
|
#endif
|