904 righe
41 KiB
MQL5
904 righe
41 KiB
MQL5
//+------------------------------------------------------------------+
|
|
//| Study.mq5 |
|
|
//| Copyright 2026, DNG |
|
|
//| https://www.mql5.com/ru/users/dng |
|
|
//+------------------------------------------------------------------+
|
|
#property copyright "Copyright 2026, DNG"
|
|
#property link "https://www.mql5.com/ru/users/dng"
|
|
#property version "1.00"
|
|
#property strict
|
|
#include "Trajectory.mqh"
|
|
//---
|
|
input group "---- AC-SRM training ----"
|
|
input datetime Start = D'2024.01.01'; //Training period start
|
|
input datetime End = D'2026.01.01'; //Training period end
|
|
input int Iterations = 200000; //Training iterations
|
|
input int EpisodeBars = 2 * StackSize; //Bars per episode
|
|
input int InpDealHorizon = 24; //Deal evaluation horizon (bars)
|
|
input double MinBalance = 50.0; //Minimum account balance
|
|
input int UpdatePolicy = 2; //Actor update interval (PolicyUpdatePeriod)
|
|
input int InpSeed = 20260921; //Replay seed base (per-owner)
|
|
input int InpUpdateTargets = 20; //Target update interval (manifest metadata)
|
|
input float InpTauT = 1.0f; //Target copy tau (1.0 = full copy)
|
|
//---
|
|
input group "---- SRM critic head ----"
|
|
input int InpRiskType = defSRM_MEAN_CVAR; //Risk measure (defSRM_*)
|
|
input float InpRiskParameter = 0.1f; //Risk parameter (alpha/lambda/nu)
|
|
input float InpMeanWeight = 0.0f; //Mean-CVaR omega
|
|
input float InpProbWeight = 1.0f; //Distribution probability weight
|
|
input float InpValueWeight = 1.0f; //Distribution value weight
|
|
input int InpQuantiles = 32; //Head quantiles (factory-fixed)
|
|
input int InpSelector = 0; //Policy SRM gradient source (0=owner 1=Q1 2=Q2)
|
|
//---
|
|
input group "---- AC-SRM production checkpoint ----"
|
|
input ENUM_OMPB_STAGE InpOMPBStage = OMPB_STAGE_BASE_POLICY; //Chain stage: keep 03 (base policy)
|
|
//---
|
|
int Epochs = 0;
|
|
CNet Actor;
|
|
CNet Q1;
|
|
CNet Q2;
|
|
|
|
CBufferFloat State;
|
|
CBufferFloat TimeState;
|
|
CBufferFloat Account;
|
|
CBufferFloat NextAccount;
|
|
CBufferFloat CurrentAction;
|
|
CBufferFloat TeacherAction;
|
|
CBufferFloat RandomAction;
|
|
CBufferFloat CriticInput;
|
|
CBufferFloat SRMValue;
|
|
CBufferFloat SRMGrad;
|
|
|
|
ulong PolicyTransitions = 0;
|
|
ulong PolicyEvents = 0;
|
|
ulong PolicySkippedInvalid = 0;
|
|
ulong CriticTransitions = 0;
|
|
ulong CriticQ1Transitions = 0;
|
|
ulong CriticQ2Transitions = 0;
|
|
ulong TeacherTransitions = 0;
|
|
ulong TeacherCriticTransitions = 0;
|
|
ulong RandomCriticTransitions = 0;
|
|
ulong CriticQ1Total = 0;
|
|
ulong CriticQ2Total = 0;
|
|
ulong UnresolvedLabels = 0;
|
|
ulong TagTP = 0;
|
|
ulong TagSL = 0;
|
|
ulong TagHorizon = 0;
|
|
double RewardSum = 0;
|
|
double CriticQ1Error = 0;
|
|
double CriticQ2Error = 0;
|
|
bool Stage03StopRequestedLogged = false;
|
|
//+-------------------------------------------------------------------+
|
|
//| Logs one user-requested Stage 03 stop without misclassifying it. |
|
|
//+-------------------------------------------------------------------+
|
|
void LogStage03StopRequested(const string phase)
|
|
{
|
|
if(Stage03StopRequestedLogged)
|
|
return;
|
|
PrintFormat("ACSRM_STAGE03_STOP_REQUESTED phase=%s", phase);
|
|
Stage03StopRequestedLogged = true;
|
|
}
|
|
//+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+
|
|
//| Creates the AC-SRM Actor and distribution-Critic architectures. Actor : Base(13)+Cross(3)+Conv(GELU 48->16)+Conv(SIGMOID 16->6). Critic: Base(19)+Cross(5)+Conv(GELU 80->16)+Base(32,None)+ ACSRM head{count=1,window_out=32,None} (factory-fixed). |
|
|
//+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+
|
|
bool CreateActorCriticDescriptions(CArrayObj *&actor_descr, CArrayObj *&critic_descr)
|
|
{
|
|
actor_descr = new CArrayObj();
|
|
critic_descr = new CArrayObj();
|
|
if(!actor_descr || !critic_descr)
|
|
{
|
|
DeleteObj(actor_descr);
|
|
DeleteObjAndFalse(critic_descr);
|
|
}
|
|
actor_descr.FreeMode(true);
|
|
critic_descr.FreeMode(true);
|
|
//--- Actor: plain scenario trunk without the D2Skill bank layer.
|
|
if(!SkillAddBase(actor_descr, AccountDescr) ||
|
|
!SkillAddCross(actor_descr, false) ||
|
|
!SkillAddConv(actor_descr, 1, (3 * EmbeddingSize), EmbeddingSize, 1, GELU) ||
|
|
!SkillAddConv(actor_descr, 1, EmbeddingSize, NActions / 2, 2, SIGMOID))
|
|
{
|
|
DeleteObj(actor_descr);
|
|
DeleteObjAndFalse(critic_descr);
|
|
}
|
|
//--- Critic: Base(19) feeds the Cross(5) trunk; the last plain Base(32,None)
|
|
//--- presents exactly 32 inputs to the factory-fixed ACSRM head.
|
|
if(!SkillAddBase(critic_descr, AccountDescr + NActions) ||
|
|
!SkillAddCross(critic_descr, true) ||
|
|
!SkillAddConv(critic_descr, 1, (5 * EmbeddingSize), EmbeddingSize, 1, GELU) ||
|
|
!SkillAddBase(critic_descr, 32))
|
|
{
|
|
DeleteObj(actor_descr);
|
|
DeleteObjAndFalse(critic_descr);
|
|
}
|
|
CLayerDescription *head = new CLayerDescription();
|
|
if(!head)
|
|
{
|
|
DeleteObj(actor_descr);
|
|
DeleteObjAndFalse(critic_descr);
|
|
}
|
|
head.type = defNeuronACSRM;
|
|
head.count = 1;
|
|
head.window_out = InpQuantiles;
|
|
head.activation = None;
|
|
head.optimization = ADAM;
|
|
head.batch = BatchSize;
|
|
if(!critic_descr.Add(head))
|
|
{
|
|
DeleteObjAndFalse(head);
|
|
DeleteObj(actor_descr);
|
|
DeleteObjAndFalse(critic_descr);
|
|
}
|
|
return(true);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Validates the AC-SRM policy shapes (no D2 banks, ACSRM heads). |
|
|
//+------------------------------------------------------------------+
|
|
bool ValidatePolicyShapeACSRM(CNet &actor, CNet &q1, CNet &q2)
|
|
{
|
|
CNeuronBaseOCL *a0 = actor.Layer(0);
|
|
CNeuronBaseOCL *a1 = actor.Layer(1);
|
|
CNeuronBaseOCL *a3 = actor.Layer(3);
|
|
if(!a0 || !a1 || !a3 || a0.getOutput().Total() != (int)AccountDescr ||
|
|
a1.Type() != defNeuronScenarioCrossAttention ||
|
|
a3.getOutput().Total() != (int)NActions)
|
|
ReturnFalse;
|
|
for(int i = 0; i < 2; i++)
|
|
{
|
|
CNet *critic = (i == 0 ? GetPointer(q1) : GetPointer(q2));
|
|
CNeuronBaseOCL *c0 = critic.Layer(0);
|
|
CNeuronBaseOCL *c3 = critic.Layer(3);
|
|
CNeuronBaseOCL *c4 = critic.Layer(4);
|
|
if(!c0 || !c3 || !c4 ||
|
|
c0.getOutput().Total() != (int)(AccountDescr + NActions) ||
|
|
c3.getOutput().Total() != 32 ||
|
|
c4.Type() != defNeuronACSRM ||
|
|
c4.getOutput().Total() != 1)
|
|
ReturnFalse;
|
|
}
|
|
return(true);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Creates fresh AC-SRM Actor + Q1 + Q2 on the Market context. |
|
|
//+------------------------------------------------------------------+
|
|
bool CreatePolicySet(void)
|
|
{
|
|
CArrayObj *actor_descr = NULL, *critic_descr = NULL;
|
|
if(!CreateActorCriticDescriptions(actor_descr, critic_descr))
|
|
{
|
|
DeleteObj(actor_descr);
|
|
DeleteObj(critic_descr);
|
|
ReturnFalse;
|
|
}
|
|
bool result = Actor.Create(actor_descr);
|
|
if(result)
|
|
result = Actor.SetOpenCLChecked(SkillMarket.GetOpenCL());
|
|
if(result)
|
|
result = Q1.Create(critic_descr);
|
|
if(result)
|
|
result = Q1.SetOpenCLChecked(SkillMarket.GetOpenCL());
|
|
if(result)
|
|
result = Q2.Create(critic_descr);
|
|
if(result)
|
|
result = Q2.SetOpenCLChecked(SkillMarket.GetOpenCL());
|
|
DeleteObj(actor_descr);
|
|
DeleteObj(critic_descr);
|
|
return(result);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Loads or creates ACSRMActor.nnw / ACSRMQ1.nnw / ACSRMQ2.nnw. |
|
|
//+------------------------------------------------------------------+
|
|
bool LoadOrCreatePolicies(void)
|
|
{
|
|
if(!SkillACRecoverTransaction(Skill_ACTOR_FILE, Skill_Q1_FILE, Skill_Q2_FILE,
|
|
Skill_AC_MANIFEST_FILE, Skill_ACTOR_NEXT_FILE,
|
|
Skill_Q1_NEXT_FILE, Skill_Q2_NEXT_FILE,
|
|
Skill_AC_MANIFEST_NEXT_FILE, Skill_ACTOR_PREVIOUS_FILE,
|
|
Skill_Q1_PREVIOUS_FILE, Skill_Q2_PREVIOUS_FILE,
|
|
Skill_AC_MANIFEST_PREVIOUS_FILE,
|
|
Skill_AC_TRANSACTION_FILE))
|
|
{
|
|
Print("Skill policy recovery=FAIL; incomplete checkpoint requires repair");
|
|
ReturnFalse;
|
|
}
|
|
const bool manifest_exists = FileIsExist(Skill_AC_MANIFEST_FILE, FILE_COMMON);
|
|
bool loaded = false;
|
|
if(manifest_exists)
|
|
{
|
|
loaded = (SkillValidateACManifestFile(true, Skill_AC_MANIFEST_FILE, false) &&
|
|
SkillLoadPolicyNet(Actor, Skill_ACTOR_FILE) &&
|
|
SkillLoadPolicyNet(Q1, Skill_Q1_FILE) &&
|
|
SkillLoadPolicyNet(Q2, Skill_Q2_FILE));
|
|
if(!loaded)
|
|
Print("Skill policy restore=FAIL; recreating incompatible policy set");
|
|
}
|
|
if(!loaded)
|
|
{
|
|
if(!CreatePolicySet())
|
|
ReturnFalse;
|
|
}
|
|
if(!SkillBindPolicyOpenCLChecked(Actor, Q1, Q2) ||
|
|
!ValidatePolicyShapeACSRM(Actor, Q1, Q2))
|
|
ReturnFalse;
|
|
Actor.TrainMode(true);
|
|
Q1.TrainMode(true);
|
|
Q2.TrainMode(true);
|
|
return(true);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Saves the Stage 03 AC-SRM policy checkpoint. |
|
|
//+------------------------------------------------------------------+
|
|
bool SavePolicies(void)
|
|
{
|
|
return(SkillSaveStage03StopCheckpoint(Actor, Q1, Q2));
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Implements PrepareHistory. |
|
|
//+------------------------------------------------------------------+
|
|
bool PrepareHistory(int &first_position, int &last_position)
|
|
{
|
|
int start = iBarShift(Symb.Name(), TimeFrame, Start);
|
|
int end = iBarShift(Symb.Name(), TimeFrame, End);
|
|
int bars = CopyRates(Symb.Name(), TimeFrame, 0, start, Rates);
|
|
if(bars <= 0 || !RSI.BufferResize(bars) || !CCI.BufferResize(bars) ||
|
|
!ATR.BufferResize(bars) || !MACD.BufferResize(bars))
|
|
ReturnFalse;
|
|
int wait = -1;
|
|
bool calculated = false;
|
|
do
|
|
{
|
|
calculated = (RSI.BarsCalculated() >= bars && CCI.BarsCalculated() >= bars &&
|
|
ATR.BarsCalculated() >= bars && MACD.BarsCalculated() >= bars);
|
|
Sleep(100);
|
|
wait++;
|
|
}
|
|
while(!calculated && wait < 100);
|
|
if(!calculated)
|
|
ReturnFalse;
|
|
RSI.Refresh();
|
|
CCI.Refresh();
|
|
ATR.Refresh();
|
|
MACD.Refresh();
|
|
if(!ArraySetAsSeries(Rates, true))
|
|
ReturnFalse;
|
|
//--- The deal outcome must be resolvable INSIDE the training window: the
|
|
//--- full horizon is reserved past the End boundary so every label
|
|
//--- (TP/SL/HORIZON) completes within [Start, End].
|
|
first_position = end + MathMax(1, SkillManifestDealHorizon);
|
|
//--- CreateBuffers(position,...,forecast) forms its state from
|
|
//--- position+NForecast and needs HistoryBars bars behind it.
|
|
last_position = start - HistoryBars - NForecast;
|
|
return(last_position > first_position);
|
|
}
|
|
//+-------------------------------------------------------------------+
|
|
//| Applies the distribution target (realized return) to one critic. |
|
|
//+-------------------------------------------------------------------+
|
|
bool UpdateCriticDistribution(CNet &critic, const double reward, const bool is_q1)
|
|
{
|
|
CNeuronBaseOCL *base = critic.Layer(3);
|
|
CNeuronBaseOCL *head = critic.Layer(4);
|
|
if(!base || !head || head.Type() != defNeuronACSRM || !MathIsValidNumber(reward) ||
|
|
base.getOutput().Total() != 32)
|
|
ReturnFalse;
|
|
CNeuronACSRM *acsrm = (CNeuronACSRM*)head;
|
|
if(!acsrm.calcDistributionTargetGradients(base, float(reward),
|
|
InpProbWeight, InpValueWeight))
|
|
ReturnFalse;
|
|
//--- Real ACSRM critic error: RMSE of the FQF centroids from the realized
|
|
//--- outcome. backPropGradient never updates getRecentAverageError(), so
|
|
//--- the old dashboard rRMSE stayed 0 forever; track the loss here.
|
|
double rmse = -1;
|
|
{
|
|
CBufferFloat *quantiles = NULL, *probabilities = NULL;
|
|
if(acsrm.GetDistributionBuffers(quantiles, probabilities) &&
|
|
quantiles != NULL && quantiles.Total() > 0 && quantiles.BufferRead())
|
|
{
|
|
double sum_sq = 0;
|
|
bool valid = true;
|
|
for(int i = 0; i < quantiles.Total(); i++)
|
|
{
|
|
const double z = double(quantiles[i]);
|
|
if(!MathIsValidNumber((float)z))
|
|
{ valid = false; break; }
|
|
sum_sq += (z - reward) * (z - reward);
|
|
}
|
|
if(valid)
|
|
rmse = MathSqrt(sum_sq / double(quantiles.Total()));
|
|
}
|
|
}
|
|
if(rmse >= 0)
|
|
{
|
|
if(is_q1)
|
|
CriticQ1Error += (rmse - CriticQ1Error) * 0.01;
|
|
else
|
|
CriticQ2Error += (rmse - CriticQ2Error) * 0.01;
|
|
}
|
|
return(critic.backPropGradient(GetPointer(SkillMarket), -1, -1, true));
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Actor policy update through the frozen critic's SRM. |
|
|
//+------------------------------------------------------------------+
|
|
int SelectPolicyCritic(const int event_index)
|
|
{
|
|
//--- Exactly ONE Critic is the gradient source per policy-update event.
|
|
//--- The default selector is balanced over Actor-update EVENTS (a hash of
|
|
//--- the event counter and the replay seed), NOT over the Owner of the
|
|
//--- training example: with a sequential position-- sampler consecutive
|
|
//--- policy events inside a stretch would share one Owner parity.
|
|
switch(InpSelector)
|
|
{
|
|
case 1: return(0);
|
|
case 2: return(1);
|
|
default: return(int((ulong(MathMax(event_index, 0)) * ulong(2654435761u) +
|
|
ulong(InpSeed)) & 1));
|
|
}
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Returns the scalar SRM value of one Critic head distribution. |
|
|
//+------------------------------------------------------------------+
|
|
bool CriticSRMValue(CNet &critic, float &value)
|
|
{
|
|
CNeuronBaseOCL *head = critic.Layer(4);
|
|
if(!head || head.Type() != defNeuronACSRM)
|
|
ReturnFalse;
|
|
CNeuronACSRM *acsrm = (CNeuronACSRM*)head;
|
|
if(!SRMValue.BufferInit(1, 0) || !SRMValue.BufferCreate(SkillMarket.GetOpenCL()) ||
|
|
!SRMValue.BufferWrite())
|
|
ReturnFalse;
|
|
if(!acsrm.SRMForward(GetPointer(SRMValue), 1, InpRiskType, InpRiskParameter, InpMeanWeight))
|
|
ReturnFalse;
|
|
if(!SRMValue.BufferRead())
|
|
ReturnFalse;
|
|
value = float(SRMValue[0]);
|
|
return(MathIsValidNumber(value));
|
|
}
|
|
//+----------------------------------------------------------------------------------------------------------------+
|
|
//| Reports whether the last SRM refusal was a distribution-contract rejection rather than an execution failure. |
|
|
//+----------------------------------------------------------------------------------------------------------------+
|
|
bool CriticLastSRMInvalid(CNet &critic)
|
|
{
|
|
CNeuronBaseOCL *head = critic.Layer(4);
|
|
if(!head || head.Type() != defNeuronACSRM)
|
|
return(false);
|
|
return(((CNeuronACSRM*)head).LastSRMInvalid());
|
|
}
|
|
//+-----------------------------------------------------------------------------------+
|
|
//| Actor policy update through the frozen critic's SRM (PolicyOutputGradientACSRM). |
|
|
//+-----------------------------------------------------------------------------------+
|
|
bool PolicyOutputGradientACSRM(CNet &actor, CNet &critic, const float upstream = 1.0f)
|
|
{
|
|
CNeuronBaseOCL *critic_head = critic.Layer(4);
|
|
CNeuronBaseOCL *critic_base = critic.Layer(0);
|
|
CNeuronBaseOCL *actor_output = actor.Layer(3);
|
|
if(!critic_head || critic_head.Type() != defNeuronACSRM || !critic_base || !actor_output ||
|
|
critic_base.getGradient().Total() != (int)(AccountDescr + NActions) ||
|
|
actor_output.getGradient().Total() != (int)NActions)
|
|
ReturnFalse;
|
|
CNeuronACSRM *acsrm = (CNeuronACSRM*)critic_head;
|
|
if(!SRMValue.BufferInit(1, 0) || !SRMValue.BufferCreate(SkillMarket.GetOpenCL()) ||
|
|
!SRMValue.BufferWrite())
|
|
ReturnFalse;
|
|
if(!acsrm.SRMForward(GetPointer(SRMValue), 1, InpRiskType, InpRiskParameter, InpMeanWeight))
|
|
ReturnFalse;
|
|
if(!SRMGrad.BufferInit(1, upstream) || !SRMGrad.BufferCreate(SkillMarket.GetOpenCL()) ||
|
|
!SRMGrad.BufferWrite())
|
|
ReturnFalse;
|
|
if(!acsrm.SRMGradient(GetPointer(SRMGrad), 1, InpRiskType, InpRiskParameter, InpMeanWeight))
|
|
ReturnFalse;
|
|
//--- Freeze critic weights while propagating the policy gradient.
|
|
if(!critic.SetWeightsUpdate(false))
|
|
ReturnFalse;
|
|
bool result = critic.backPropGradient(GetPointer(SkillMarket), -1, -1, true);
|
|
if(!critic.SetWeightsUpdate(true))
|
|
result = false;
|
|
CBufferFloat account_gradient;
|
|
if(!(result &&
|
|
account_gradient.BufferInit(AccountDescr, 0) &&
|
|
(account_gradient.GetIndex() >= 0 || account_gradient.BufferCreate(SkillMarket.GetOpenCL())) &&
|
|
SkillDevice.Split2(GetPointer(account_gradient), actor_output.getGradient(),
|
|
critic_base.getGradient(), AccountDescr, NActions) &&
|
|
ApplyActorSigmoidDerivative(actor_output)))
|
|
result = false;
|
|
return(result);
|
|
}
|
|
//+---------------------------------------------------------------------------------------------------------------------+
|
|
//| Accumulates one supervised signal (teacher/random) into the actor output-layer gradient: (t - a) .* a .* (1 - a). |
|
|
//+---------------------------------------------------------------------------------------------------------------------+
|
|
bool AddActorSupervisedGradient(CNet &actor, CBufferFloat *target)
|
|
{
|
|
CNeuronBaseOCL *actor_output = actor.Layer(3);
|
|
if(!actor_output || !target || target.Total() != (int)NActions ||
|
|
actor_output.getOutput().Total() != (int)NActions ||
|
|
actor_output.getGradient().Total() != (int)NActions || target.GetIndex() < 0 ||
|
|
actor_output.getOutput().GetIndex() < 0 || actor_output.getGradient().GetIndex() < 0)
|
|
ReturnFalse;
|
|
if(!ActorDerivOneMinus.BufferInit(NActions, 0) ||
|
|
!ActorDerivOneMinus.BufferCreate(SkillMarket.GetOpenCL()) ||
|
|
!ActorDerivOneMinus.BufferWrite() ||
|
|
!ActorGradScratch.BufferInit(NActions, 0) ||
|
|
!ActorGradScratch.BufferCreate(SkillMarket.GetOpenCL()) ||
|
|
!ActorGradScratch.BufferWrite() ||
|
|
!SkillDevice.Bind(SkillMarket.GetOpenCL()))
|
|
ReturnFalse;
|
|
//--- diff = t - a; scaled = diff * a * (1 - a); accumulate on device.
|
|
//--- This reproduces the library's calcOutputGradients for a SIGMOID
|
|
//--- output exactly, but pre-update and summed before ONE weight update.
|
|
if(!SkillDevice.Subtract(target, actor_output.getOutput(), GetPointer(ActorDerivOneMinus), NActions) ||
|
|
!SkillDevice.DeActivationOnDevice(actor_output.getOutput(), GetPointer(ActorGradScratch),
|
|
GetPointer(ActorDerivOneMinus), SIGMOID) ||
|
|
!SkillDevice.Add(actor_output.getGradient(), GetPointer(ActorGradScratch),
|
|
actor_output.getGradient(), NActions))
|
|
ReturnFalse;
|
|
return(true);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Implements TrainTransition. |
|
|
//+------------------------------------------------------------------+
|
|
bool TrainTransition(const int position, const int iteration, bool &terminal)
|
|
{
|
|
terminal = false;
|
|
const bool trace_td = (iteration < 24 || iteration % 10000 == 0 || iteration >= Iterations - 1);
|
|
if(!SkillForwardForecast(position, GetPointer(State), GetPointer(TimeState)))
|
|
{ Print("TrainTransition stage=current_forecast"); ReturnFalse; }
|
|
double reward = 0;
|
|
if(!Actor.feedForward(GetPointer(Account), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!ReadAction(Actor, GetPointer(CurrentAction)) ||
|
|
!AdvanceAccount(GetPointer(Account), GetPointer(CurrentAction), position, MinBalance,
|
|
GetPointer(NextAccount), reward, terminal))
|
|
{ Print("TrainTransition stage=account_transition"); ReturnFalse; }
|
|
const double buy_lot = MathMax(0.0, double(CurrentAction[0] - CurrentAction[3]));
|
|
const double sell_lot = MathMax(0.0, double(CurrentAction[3] - CurrentAction[0]));
|
|
if(trace_td)
|
|
PrintFormat("ACSRM action iteration=%d balance=%.2f buy_lot=%.8f buy_tp=%.8f buy_sl=%.8f " +
|
|
"sell_lot=%.8f sell_tp=%.8f sell_sl=%.8f",
|
|
iteration, MathMax(0.0, double(Account[0]) * EtalonBalance),
|
|
buy_lot, double(CurrentAction[1]), double(CurrentAction[2]),
|
|
sell_lot, double(CurrentAction[4]), double(CurrentAction[5]));
|
|
if(trace_td)
|
|
PrintFormat("ACSRM return iteration=%d reward=%.8f balance=%.2f terminal=%s",
|
|
iteration, reward, MathMax(0.0, double(NextAccount[0]) * EtalonBalance),
|
|
(terminal ? "true" : "false"));
|
|
//--- Critic input is built from the live device activations.
|
|
CNeuronBaseOCL *context_layer = Actor.Layer(0);
|
|
CNeuronBaseOCL *actor_layer = Actor.Layer(3);
|
|
if(!context_layer || !actor_layer ||
|
|
!BuildCriticInput(context_layer.getOutput(), actor_layer.getOutput(), GetPointer(CriticInput)))
|
|
{ Print("TrainTransition stage=critic_input"); ReturnFalse; }
|
|
if(!Q1.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!Q2.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1))
|
|
{ Print("TrainTransition stage=critic_forward"); ReturnFalse; }
|
|
//--- Offline target: the COMPLETED result of the specific deal fixed at this
|
|
//--- bar over the configured horizon (TP first / SL first / HORIZON outcome),
|
|
//--- discounted by elapsed bars. Realized future bars may label the deal but
|
|
//--- never enter the decision state (SkillForwardForecast already shifts the
|
|
//--- window to the strictly-past bars). It is a finished result and needs no
|
|
//--- target bootstrap.
|
|
int deal_tag = 0;
|
|
const double deal_y = EvaluateAction(GetPointer(CurrentAction),
|
|
MathMax(0.0, double(Account[0]) * EtalonBalance),
|
|
(uint)position, InpDealHorizon, deal_tag);
|
|
if(!MathIsValidNumber(deal_y))
|
|
{ Print("TrainTransition stage=deal_label"); ReturnFalse; }
|
|
if(trace_td)
|
|
PrintFormat("ACSRM offline iteration=%d outcome=%.8f tag=%d advance_reward=%.8f horizon=%d",
|
|
iteration, deal_y, deal_tag, reward, InpDealHorizon);
|
|
//--- Offline Owner separation: the example id (position,slot,seed) fixes one
|
|
//--- stable Owner-Critic; re-reading the same record keeps its label there.
|
|
const int owner = SkillObservationOwner(position, 0, InpSeed);
|
|
if(deal_tag == 0)
|
|
UnresolvedLabels++;
|
|
else
|
|
{
|
|
CNet *owner_critic = (owner == 0 ? GetPointer(Q1) : GetPointer(Q2));
|
|
if(!owner_critic.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!UpdateCriticDistribution(owner_critic, deal_y, (owner_critic == GetPointer(Q1))))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("critic_distribution_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=critic_distribution_backward");
|
|
ReturnFalse;
|
|
}
|
|
CriticTransitions++;
|
|
if(owner_critic == GetPointer(Q1))
|
|
{
|
|
CriticQ1Transitions++;
|
|
CriticQ1Total++;
|
|
}
|
|
else
|
|
{
|
|
CriticQ2Transitions++;
|
|
CriticQ2Total++;
|
|
}
|
|
RewardSum += deal_y;
|
|
if(deal_tag == 1)
|
|
TagTP++;
|
|
else
|
|
if(deal_tag == 2)
|
|
TagSL++;
|
|
else
|
|
TagHorizon++;
|
|
}
|
|
//--- Actor multi-objective update. Policy, teacher and random signals are
|
|
//--- computed on the SAME forward activations and the SAME pre-update
|
|
//--- weights and ACCUMULATED into the output-layer gradient; ONE
|
|
//--- backPropGradient applies them together. No weight change happens
|
|
//--- between the contributions so credit assignment stays consistent.
|
|
bool actor_update_needed = false;
|
|
bool actor_update_blocked = false;
|
|
//--- Gradient hygiene: clear the Actor output gradient once BEFORE the
|
|
//--- current observation accumulates its contributions. A deferred policy
|
|
//--- event (UpdatePolicy > 1) would otherwise Add teacher/random on top of
|
|
//--- the stale gradient the previous combined update consumed, since Split2
|
|
//--- only overwrites the gradient on the policy event.
|
|
if(actor_layer.getGradient().Total() != (int)NActions ||
|
|
!actor_layer.getGradient().Fill(0))
|
|
{ Print("TrainTransition stage=actor_gradient_reset"); ReturnFalse; }
|
|
//--- Policy gate: a RESOLVABLE deal label (tag != 0). Requiring broker
|
|
//--- executability here would give the cold-start Actor no gradient while
|
|
//--- its lots are below LotsMin(); with the min-lot floor inside
|
|
//--- CheckAction the gradient direction exists from the first iteration.
|
|
const bool resolvable_action = (deal_tag != 0);
|
|
if(iteration > 0 && UpdatePolicy > 0 && iteration % UpdatePolicy == 0)
|
|
{
|
|
if(resolvable_action)
|
|
{
|
|
//--- Fresh Critic activations on the same observation AFTER the Critic
|
|
//--- weight update: the policy gradient must not mix pre-update
|
|
//--- activations with post-update weights.
|
|
if(!Q1.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!Q2.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1))
|
|
{ Print("TrainTransition stage=policy_critic_forward"); ReturnFalse; }
|
|
const int pick = SelectPolicyCritic(int(PolicyEvents++));
|
|
CNet *policy_critic = (pick == 0 ? GetPointer(Q1) : GetPointer(Q2));
|
|
float policy_srm = 0.0f;
|
|
if(!CriticSRMValue(*policy_critic, policy_srm))
|
|
{
|
|
if(CriticLastSRMInvalid(*policy_critic))
|
|
{
|
|
PolicySkippedInvalid++;
|
|
actor_update_blocked = true;
|
|
if(trace_td)
|
|
PrintFormat("ACSRM policy skipped iteration=%d reason=invalid_srm_distribution", iteration);
|
|
}
|
|
else
|
|
{ Print("TrainTransition stage=policy_critic_srm"); ReturnFalse; }
|
|
}
|
|
else
|
|
{
|
|
const bool policy_ok = PolicyOutputGradientACSRM(Actor, *policy_critic, 1.0f);
|
|
if(!policy_ok)
|
|
{
|
|
if(CriticLastSRMInvalid(*policy_critic))
|
|
{
|
|
PolicySkippedInvalid++;
|
|
actor_update_blocked = true;
|
|
if(trace_td)
|
|
PrintFormat("ACSRM policy skipped iteration=%d reason=invalid_srm_gradient", iteration);
|
|
}
|
|
else
|
|
{ Print("TrainTransition stage=policy_backward"); ReturnFalse; }
|
|
}
|
|
else
|
|
{
|
|
PolicyTransitions++;
|
|
actor_update_needed = true;
|
|
}
|
|
}
|
|
}
|
|
else
|
|
PolicySkippedInvalid++;
|
|
}
|
|
//--- Restored original-project Actor signals: a profitable realized-future
|
|
//--- teacher and a profitable random executable action re-enter the action
|
|
//--- manifold; their deals also expand Owner-Critic coverage. Only
|
|
//--- RESOLVABLE labels (tag != 0) train anything: with the SL sign fixed
|
|
//--- the reward > 0 gate now stands for a genuine TP profit.
|
|
double teacher_reward = 0;
|
|
int teacher_tag = 0;
|
|
if(!BuildTeacherAction(position, GetPointer(Account), GetPointer(TeacherAction),
|
|
teacher_reward, InpDealHorizon, teacher_tag))
|
|
{ Print("TrainTransition stage=teacher_action"); ReturnFalse; }
|
|
if(trace_td)
|
|
PrintFormat("ACSRM teacher iteration=%d reward=%.8f tag=%d", iteration, teacher_reward, teacher_tag);
|
|
if(teacher_reward > 0.0 && teacher_tag != 0)
|
|
{
|
|
if(!AddActorSupervisedGradient(Actor, GetPointer(TeacherAction)))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("teacher_actor_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=teacher_actor_backward");
|
|
ReturnFalse;
|
|
}
|
|
TeacherTransitions++;
|
|
actor_update_needed = true;
|
|
}
|
|
if(teacher_tag == 0)
|
|
UnresolvedLabels++;
|
|
else
|
|
{
|
|
const int teacher_owner = SkillObservationOwner(position, 1, InpSeed);
|
|
CNet *teacher_critic = (teacher_owner == 0 ? GetPointer(Q1) : GetPointer(Q2));
|
|
if(!BuildCriticInput(context_layer.getOutput(), GetPointer(TeacherAction), GetPointer(CriticInput)) ||
|
|
!teacher_critic.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!UpdateCriticDistribution(teacher_critic, teacher_reward, (teacher_critic == GetPointer(Q1))))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("teacher_critic_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=teacher_critic_backward");
|
|
ReturnFalse;
|
|
}
|
|
TeacherCriticTransitions++;
|
|
if(teacher_critic == GetPointer(Q1))
|
|
CriticQ1Total++;
|
|
else
|
|
CriticQ2Total++;
|
|
}
|
|
double random_reward = 0;
|
|
int random_tag = 0;
|
|
if(!BuildRandomAction(position, GetPointer(Account), GetPointer(RandomAction),
|
|
random_reward, InpDealHorizon, random_tag))
|
|
{ Print("TrainTransition stage=random_action"); ReturnFalse; }
|
|
if(trace_td)
|
|
PrintFormat("ACSRM random iteration=%d reward=%.8f tag=%d", iteration, random_reward, random_tag);
|
|
if(random_reward > 0.0 && random_tag != 0)
|
|
{
|
|
if(!AddActorSupervisedGradient(Actor, GetPointer(RandomAction)))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("random_actor_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=random_actor_backward");
|
|
ReturnFalse;
|
|
}
|
|
actor_update_needed = true;
|
|
}
|
|
if(random_tag == 0)
|
|
UnresolvedLabels++;
|
|
else
|
|
{
|
|
const int random_owner = SkillObservationOwner(position, 2, InpSeed);
|
|
CNet *random_critic = (random_owner == 0 ? GetPointer(Q1) : GetPointer(Q2));
|
|
if(!BuildCriticInput(context_layer.getOutput(), GetPointer(RandomAction), GetPointer(CriticInput)) ||
|
|
!random_critic.feedForward(GetPointer(CriticInput), 1, false, GetPointer(SkillMarket), -1) ||
|
|
!UpdateCriticDistribution(random_critic, random_reward, (random_critic == GetPointer(Q1))))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("random_critic_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=random_critic_backward");
|
|
ReturnFalse;
|
|
}
|
|
RandomCriticTransitions++;
|
|
if(random_critic == GetPointer(Q1))
|
|
CriticQ1Total++;
|
|
else
|
|
CriticQ2Total++;
|
|
}
|
|
//--- ONE coherent Actor weight update with the combined gradient (if any).
|
|
//--- An invalid SRM distribution blocks the whole Actor update for this
|
|
//--- observation, matching srm_invalid_policy=skip_actor_update.
|
|
if(actor_update_needed && !actor_update_blocked &&
|
|
!Actor.backPropGradient(GetPointer(SkillMarket), -1, -1, true))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("actor_combined_backward");
|
|
return(false);
|
|
}
|
|
Print("TrainTransition stage=actor_combined_backward");
|
|
ReturnFalse;
|
|
}
|
|
return(true);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Implements TrainACSRMActorCritic. |
|
|
//+------------------------------------------------------------------+
|
|
void TrainACSRMActorCritic(void)
|
|
{
|
|
Stage03StopRequestedLogged = false;
|
|
int first = 0, last = 0;
|
|
if(!PrepareHistory(first, last))
|
|
{ PrintFormat("%s -> %d history unavailable", __FUNCTION__, __LINE__); return; }
|
|
uint shown = GetTickCount();
|
|
int completed_iterations = 0;
|
|
int failed_iteration = -1;
|
|
int failed_position = -1;
|
|
int position = last;
|
|
int episode = 0;
|
|
PolicyTransitions = 0;
|
|
PolicyEvents = 0;
|
|
PolicySkippedInvalid = 0;
|
|
CriticTransitions = 0;
|
|
CriticQ1Transitions = 0;
|
|
CriticQ2Transitions = 0;
|
|
TeacherTransitions = 0;
|
|
TeacherCriticTransitions = 0;
|
|
RandomCriticTransitions = 0;
|
|
CriticQ1Total = 0;
|
|
CriticQ2Total = 0;
|
|
UnresolvedLabels = 0;
|
|
TagTP = 0;
|
|
TagSL = 0;
|
|
TagHorizon = 0;
|
|
RewardSum = 0;
|
|
CriticQ1Error = 0;
|
|
CriticQ2Error = 0;
|
|
for(int iteration = 0; iteration < Iterations && !IsStopped(); iteration++)
|
|
{
|
|
//--- Fresh episode: seed the sampler from the observation Owner.
|
|
if(episode == 0)
|
|
{
|
|
if(!SkillForwardForecast(position, GetPointer(State), GetPointer(TimeState)))
|
|
break;
|
|
MathSrand(InpSeed + position);
|
|
const vector<float> sampled = SampleAccount(GetPointer(State), Rates[position].time,
|
|
EtalonBalance, MinBalance);
|
|
if(sampled.Size() != AccountDescr || !Account.AssignArray(sampled) ||
|
|
(Account.GetIndex() < 0 && !Account.BufferCreate(SkillMarket.GetOpenCL())) ||
|
|
!Account.BufferWrite() || !Actor.Clear())
|
|
break;
|
|
}
|
|
bool terminal = false;
|
|
if(!TrainTransition(position, iteration, terminal))
|
|
{
|
|
if(IsStopped())
|
|
{
|
|
LogStage03StopRequested("train_transition");
|
|
break;
|
|
}
|
|
failed_iteration = iteration;
|
|
failed_position = position;
|
|
break;
|
|
}
|
|
completed_iterations++;
|
|
if((NextAccount.GetIndex() >= 0 && !NextAccount.BufferRead()) ||
|
|
!Account.AssignArray(GetPointer(NextAccount)) ||
|
|
(Account.GetIndex() >= 0 && !Account.BufferWrite()))
|
|
{
|
|
failed_iteration = iteration;
|
|
failed_position = position;
|
|
break;
|
|
}
|
|
position--;
|
|
episode++;
|
|
//--- Close the episode on range exhaustion, its bar limit or a
|
|
//--- terminal account state.
|
|
if(position < first || episode >= MathMax(1, EpisodeBars) || terminal)
|
|
{
|
|
if(iteration % 10000 == 0)
|
|
PrintFormat("ACSRM episode closed iteration=%d episode=%d position=%d terminal=%s",
|
|
iteration, episode, position, (terminal ? "true" : "false"));
|
|
if(position < first)
|
|
position = last;
|
|
episode = 0;
|
|
}
|
|
//--- Refresh the progress panel without delaying training updates.
|
|
if(GetTickCount() - shown > 500)
|
|
{
|
|
|
|
|
|
|
|
|
|
const double r_mean = (CriticTransitions > 0 ? RewardSum / double(CriticTransitions) : 0.0);
|
|
Comment(StringFormat("ACSRM AC %6.2f%% Q1 err %.8f Q2 err %.8f critic %I64u " +
|
|
"policy %I64u skip %I64u Rmean %.4f",
|
|
100.0 * iteration / MathMax(Iterations, 1), CriticQ1Error, CriticQ2Error,
|
|
CriticTransitions, PolicyTransitions, PolicySkippedInvalid,
|
|
r_mean));
|
|
shown = GetTickCount();
|
|
}
|
|
}
|
|
if(IsStopped())
|
|
{
|
|
//--- A stop can land between parameter updates; never publish a partial step.
|
|
LogStage03StopRequested("training_loop");
|
|
Comment("");
|
|
PrintFormat("ACSRM_STAGE03_STOP_NO_PUBLISH completed=%d/%d critic=%I64u q1=%I64u q2=%I64u policy=%I64u",
|
|
completed_iterations, Iterations, CriticTransitions,
|
|
CriticQ1Transitions, CriticQ2Transitions, PolicyTransitions);
|
|
return;
|
|
}
|
|
Comment("");
|
|
//--- Reject publication unless every requested iteration completed safely.
|
|
if(completed_iterations != Iterations)
|
|
{
|
|
if(failed_iteration >= 0)
|
|
PrintFormat("%s -> %d publication aborted: TrainTransition failed iteration=%d position=%d completed=%d/%d",
|
|
__FUNCTION__, __LINE__, failed_iteration, failed_position,
|
|
completed_iterations, Iterations);
|
|
else
|
|
PrintFormat("%s -> %d publication aborted: stop requested completed=%d/%d",
|
|
__FUNCTION__, __LINE__, completed_iterations, Iterations);
|
|
return;
|
|
}
|
|
if(!SavePolicies())
|
|
PrintFormat("%s -> %d save failed", __FUNCTION__, __LINE__);
|
|
else
|
|
PrintFormat("ACSRM_STUDY_PASS iterations=%d critic=%I64u q1=%I64u q2=%I64u " +
|
|
"q1_total=%I64u q2_total=%I64u policy=%I64u skip=%I64u teacher=%I64u " +
|
|
"teacher_critic=%I64u random_critic=%I64u unresolved=%I64u " +
|
|
"tags_tp=%I64u tags_sl=%I64u tags_horizon=%I64u " +
|
|
"q1_err=%.8f q2_err=%.8f published=canonical",
|
|
completed_iterations, CriticTransitions, CriticQ1Transitions, CriticQ2Transitions,
|
|
CriticQ1Total, CriticQ2Total,
|
|
PolicyTransitions, PolicySkippedInvalid, TeacherTransitions,
|
|
TeacherCriticTransitions, RandomCriticTransitions, UnresolvedLabels,
|
|
TagTP, TagSL, TagHorizon, CriticQ1Error, CriticQ2Error);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Initializes the Stage 03 BasePolicy Expert. |
|
|
//+------------------------------------------------------------------+
|
|
int OnInit()
|
|
{
|
|
ResetLastError();
|
|
//--- Pin the deal-evaluation horizon for manifest writer/validator symmetry.
|
|
SkillManifestDealHorizon = MathMax(1, InpDealHorizon);
|
|
SkillManifestTargetTau = double(InpTauT);
|
|
SkillManifestTargetUpdatePeriod = InpUpdateTargets;
|
|
if(InpSelector < 0 || InpSelector > 2 ||
|
|
InpUpdateTargets < 0 || InpTauT < 0.0f || InpTauT > 1.0f)
|
|
{
|
|
PrintFormat("ACSRM input validation failed at line %d error=%d",
|
|
__LINE__, GetLastError());
|
|
return(INIT_FAILED);
|
|
}
|
|
//--- Confirm Stage 03 and that the frozen Forecast checkpoint is intact.
|
|
if(InpOMPBStage != OMPB_STAGE_BASE_POLICY ||
|
|
!SkillInitIndicators() || !SkillLoadForecastInference() ||
|
|
!SkillConfigureProductionACSRMCheckpoint() ||
|
|
!SkillValidateProductionACSRMCheckpoint() ||
|
|
!SkillCaptureProductionACSRMFingerprints() || !LoadOrCreatePolicies() ||
|
|
!SkillVerifyFrozenForecastExact() ||
|
|
!SkillVerifyProductionACSRMFingerprints())
|
|
{
|
|
PrintFormat("ACSRM Actor-Critic initialization failed at line %d error=%d",
|
|
__LINE__, GetLastError());
|
|
return(INIT_FAILED);
|
|
}
|
|
if(!EventSetMillisecondTimer(1))
|
|
{
|
|
PrintFormat("ACSRM Actor-Critic timer initialization failed at line %d error=%d",
|
|
__LINE__, GetLastError());
|
|
return(INIT_FAILED);
|
|
}
|
|
//--- Finalize initialization only after the Stage 03 timer is armed.
|
|
return(INIT_SUCCEEDED);
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Function OnDeinit. |
|
|
//+------------------------------------------------------------------+
|
|
void OnDeinit(const int reason)
|
|
{
|
|
EventKillTimer();
|
|
//--- Only a completed explicit training loop persists all three artifacts
|
|
//--- and writes the hash manifest last. Deinit must not create a mixed
|
|
//--- generation.
|
|
if(SkillForecast != NULL && !SkillVerifyFrozenForecastExact())
|
|
PrintFormat("%s -> %d forecast mutation", __FUNCTION__, __LINE__);
|
|
if(SkillProductionSignatureReady && !SkillVerifyProductionACSRMFingerprints())
|
|
PrintFormat("%s -> %d ACSRM production signature mutation", __FUNCTION__, __LINE__);
|
|
SkillForecast = NULL;
|
|
}
|
|
//+------------------------------------------------------------------+
|
|
//| Runs the one-shot training lifecycle. |
|
|
//+------------------------------------------------------------------+
|
|
void OnTimer(void)
|
|
{
|
|
TrainACSRMActorCritic();
|
|
ExpertRemove();
|
|
}
|
|
//+------------------------------------------------------------------
|
|
//+---
|