2026-08-01 11:27:28 -04:00
//+------------------------------------------------------------------+
//| NetPersistence.mqh |
//| |
//| CNet::Save / CNet::Load - the .nnw format. |
//| |
//| Included from AI\Network.mqh AFTER every class declaration - |
//| bodies only, no declarations. Relocation is behaviour-neutral by |
//| construction: nothing here is reachable until Network.mqh ends. |
//+------------------------------------------------------------------+
# ifndef WARRIOR_AI_IMPL_NETPERSISTENCE_MQH
# define WARRIOR_AI_IMPL_NETPERSISTENCE_MQH
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNet : : Save ( string file_name , double error , double undefine , double forecast , datetime time , bool common , long era , bool trainingComplete , const double & indicatorParams [ ] )
{
//--- only the shared FILE_COMMON production weights are protected from being overwritten by
//--- backtest/optimization noise - a LOCAL (common=false) file is exactly what the tester's
//--- per-agent cross-pass weight cache uses (see CExpertSignalAIBase::InitNeuralNetwork), and
//--- must be allowed to write even inside the tester/optimizer.
if ( common & & ( MQLInfoInteger ( MQL_OPTIMIZATION ) | | MQLInfoInteger ( MQL_TESTER ) | | MQLInfoInteger ( MQL_FORWARD ) ) )
return true ;
if ( file_name = = NULL )
return false ;
//--- Never persist a layerless network. FileOpen(FILE_WRITE) below TRUNCATES on open, so writing an
//--- empty net here would replace a good ~18MB model on disk with a ~330-byte header-only stub (0
//--- layers) - silent, total loss of trained weights. This guard is checked BEFORE FileOpen so a bad
//--- in-memory state (e.g. a failed checkpoint restore that emptied the layers) can't clobber the file.
if ( CheckPointer ( layers ) = = POINTER_INVALID | | layers . Total ( ) = = 0 )
{
Print ( " CNet::Save: refusing to write " , file_name , " - network has 0 layers (would overwrite a valid model with an empty stub) " ) ;
return false ;
}
//--- ATOMIC SAVE (crash-safe) - see System\AtomicFile.mqh for why every write here is staged through a
//--- temp file. Short version: FileOpen(FILE_WRITE) truncates on open, so writing straight to file_name
//--- would zero the good model the instant the save starts, and an interrupted save then leaves a
//--- truncated .nnw that the next load rebuilds from era 0 (the recurring data loss).
int commonFlag = ( common ? FILE_COMMON : 0 ) ;
string tmpName = " " ;
int handle = AtomicWriteBegin ( file_name , commonFlag , tmpName ) ;
if ( handle = = INVALID_HANDLE )
return false ;
//---
bool ok = true ;
if ( FileWriteDouble ( handle , error ) < = 0 | | FileWriteDouble ( handle , undefine ) < = 0 | | FileWriteDouble ( handle , forecast ) < = 0 | | FileWriteLong ( handle , ( long ) time ) < = 0 | | FileWriteLong ( handle , era ) < = 0 | |
FileWriteInteger ( handle , trainingComplete ? 1 : 0 ) < = 0 )
ok = false ;
//--- AutoTuneIndicators "winning" AD indicator params, same append-only pattern as the era/
//--- trainingComplete fields above: count-prefixed so older readers can still stop before this block
int paramsCount = ArraySize ( indicatorParams ) ;
if ( ok & & FileWriteInteger ( handle , paramsCount ) < = 0 )
ok = false ;
for ( int p = 0 ; ok & & p < paramsCount ; p + + )
if ( FileWriteDouble ( handle , indicatorParams [ p ] ) < = 0 )
ok = false ;
if ( ok & & ! layers . Save ( handle ) )
ok = false ;
return AtomicWriteEnd ( handle , file_name , tmpName , commonFlag , ok , " CNet::Save " ) ;
}
//+------------------------------------------------------------------+
//| |
//+------------------------------------------------------------------+
bool CNet : : Load ( string file_name , double & error , double & undefine , double & forecast , datetime & time , bool common , long & era , bool & trainingComplete , double & indicatorParams [ ] , bool quiet )
{
//--- see the matching comment in Save() - only shared FILE_COMMON production weights are blocked
if ( common & & ( MQLInfoInteger ( MQL_OPTIMIZATION ) | | MQLInfoInteger ( MQL_TESTER ) | | MQLInfoInteger ( MQL_FORWARD ) ) )
return false ;
//---
if ( file_name = = NULL )
return false ;
//---
if ( ! quiet )
Print ( file_name ) ;
//--- FILE_SHARE_READ|FILE_SHARE_WRITE: this is a READ - it never needs exclusive access, and demanding
//--- it (the default) makes the open fail with 5004 whenever any other process holds the file, which is
//--- the normal state of a deployed model while a live chart runs this EA. See CopySharedFile().
int handle = FileOpen ( file_name , ( common ? FILE_COMMON : 0 ) | FILE_BIN | FILE_READ | FILE_SHARE_READ | FILE_SHARE_WRITE ) ;
if ( handle = = INVALID_HANDLE )
{
2026-08-02 01:09:18 -04:00
//--- Every OTHER rejection below prints why, and so does this one - a real sharing violation from a
//--- concurrent writer (e.g. a live chart's atomic Save() renaming this same file mid-read) must not
//--- look identical to "no file yet" in the journal, or it falls through silently to an untrained net.
//--- Ask whether the file is actually THERE rather than making the reader guess. FileOpen returns
//--- 5004 for both "no such file" and "locked by another process", so the old wording hedged across
//--- both - and on a first run for a new configuration it printed a lock warning, five times, for
//--- the entirely normal state of having no saved model yet. That noise sat at the top of the
//--- 2026-08-02 journal and drew the first round of diagnosis to the wrong subsystem.
2026-08-01 11:27:28 -04:00
if ( ! quiet )
2026-08-02 01:09:18 -04:00
{
int err = GetLastError ( ) ;
if ( FileIsExist ( file_name , ( common ? FILE_COMMON : 0 ) ) )
Print ( " CNet::Load: " + file_name + " - the file EXISTS but FileOpen failed, error " +
IntegerToString ( err ) + " . That is a sharing violation, not a missing model - another " +
" process is holding it (e.g. a live chart's atomic Save() renaming it mid-read). " ) ;
else
Print ( " CNet::Load: " + file_name + " - no such file (error " + IntegerToString ( err ) +
" ). Normal for a configuration that has never been trained; the caller starts from era 0. " ) ;
}
2026-08-01 11:27:28 -04:00
return false ;
}
//--- Capture the on-disk size up front: it discriminates the two failure modes a rejected read can have.
//--- A healthy multi-layer model is multi-MB; a size of only a few hundred bytes/KB = a TRUNCATED file
//--- (interrupted save). A full-size file that STILL fails to load layers = a backend ALLOCATION failure
//--- (e.g. the CPU-DLL still holding a prior net's tensors after a skipped teardown), NOT truncation.
ulong fileSizeBytes = FileSize ( handle ) ;
//---
error = FileReadDouble ( handle ) ;
undefine = FileReadDouble ( handle ) ;
forecast = FileReadDouble ( handle ) ;
time = ( datetime ) FileReadLong ( handle ) ;
era = FileReadLong ( handle ) ;
// Older save files predate this field - FileReadInteger returns 0 past EOF,
// which correctly defaults to "not complete" so a resumed run keeps training
// instead of silently trusting an unfinished/unknown model as done.
trainingComplete = ( FileReadInteger ( handle ) ! = 0 ) ;
// Older save files predate this block too - FileReadInteger returns 0 past EOF, which
// correctly yields an empty indicatorParams (nothing to restore) instead of misreading weights.
ArrayFree ( indicatorParams ) ;
int paramsCount = FileReadInteger ( handle ) ;
if ( paramsCount > 0 )
{
ArrayResize ( indicatorParams , paramsCount ) ;
for ( int p = 0 ; p < paramsCount ; p + + )
indicatorParams [ p ] = FileReadDouble ( handle ) ;
}
//---
if ( CheckPointer ( layers ) ! = POINTER_INVALID )
layers . Clear ( ) ;
else
layers = new CArrayLayer ( ) ;
int i = 0 , num ;
//---
if ( ! InitOpenCL ( ) )
InitDirectML ( ) ;
//--- check
//--- read and check start marker - 0xFFFFFFFFFFFFFFFF
//--- DIAGNOSTIC: every failure below returns a bare false, and the caller can only surface GetLastError()
//--- which on a no-GPU/CPU-DLL box is polluted with the harmless 5100 from the OpenCL probe above - so a
//--- genuinely corrupt/partial/incompatible file looked identical to a missing one for several sessions of
//--- "restart -> era 0" hunting. Name the exact failure so the next restart is diagnosable at a glance.
long temp = FileReadLong ( handle ) ;
if ( temp ! = -1 )
{
FileClose ( handle ) ;
if ( ! quiet )
Print ( " CNet::Load: " + file_name + " - REJECTED: bad start marker (read " + IntegerToString ( temp ) +
" , expected -1) => the header is corrupt or the file was truncated by an interrupted save. This is NOT a compute/OpenCL problem. " ) ;
return ( false ) ;
}
//--- read and check array type
int savedType = FileReadInteger ( handle , INT_VALUE ) ;
if ( savedType ! = layers . Type ( ) )
{
FileClose ( handle ) ;
if ( ! quiet )
Print ( " CNet::Load: " + file_name + " - REJECTED: layer-array type mismatch (read " + IntegerToString ( savedType ) +
" , expected " + IntegerToString ( layers . Type ( ) ) + " ) => file format/version incompatible. This is NOT a compute/OpenCL problem. " ) ;
return ( false ) ;
}
//--- read array length
num = FileReadInteger ( handle , INT_VALUE ) ;
//--- read array
int failedLayer = -1 ; // index of the first layer that failed to load, or -1 if all loaded
if ( num ! = 0 )
{
for ( i = 0 ; i < num ; i + + )
{
//--- create new element
CLayer * Layer = new CLayer ( 0 , handle , opencl , directml ) ;
if ( CheckPointer ( Layer ) = = POINTER_INVALID )
{
failedLayer = i ;
break ;
}
//--- On a partial/corrupt file, Layer.Load() or layers.Add() can fail after the CLayer was
//--- already allocated. Deleting it before breaking prevents the orphaned-CLayer leak MT5 reports
//--- at unload ("1 object of class 'CLayer'") - the layer was never added to `layers`, so the
//--- CNet destructor (which frees `layers`) can't reclaim it.
if ( ! Layer . Load ( handle ) )
{
delete Layer ;
failedLayer = i ;
break ;
}
if ( ! layers . Add ( Layer ) )
{
delete Layer ;
failedLayer = i ;
break ;
}
}
}
FileClose ( handle ) ;
//--- result: a 0-layer file is not a usable model (it's the empty stub the Save guard now prevents, but
//--- older stubs may still be on disk) - report failure so the caller rebuilds/retrains instead of
//--- running inference on a layerless network.
if ( num = = 0 )
{
if ( ! quiet )
Print ( " CNet::Load: " + file_name + " - REJECTED: 0-layer stub (empty model, header says " +
IntegerToString ( num ) + " layers). An older empty-save stub; the current Save guard prevents these. Use reset-weights to start clean. NOT a compute/OpenCL problem. " ) ;
return ( false ) ;
}
if ( failedLayer > = 0 | | layers . Total ( ) ! = num )
{
if ( ! quiet )
{
string backend = ( CheckPointer ( opencl ) ! = POINTER_INVALID ? " OpenCL " : ( CheckPointer ( directml ) ! = POINTER_INVALID ? " CPU-DLL " : " pure-MQL5 " ) ) ;
//--- Which failure mode this is depends on WHERE it stopped, not just on the file size:
//--- - failed at layer 0 with a full-size file: nothing was read yet, so the file cannot be the
//--- cause. This is a CODE fault in the read path - historically CLayer::CreateElement losing
//--- its virtual override of CArrayObj::CreateElement (see the note on that method), which
//--- made CArrayObj::Load() hit the base's `return(false)` stub on the very first neuron.
//--- - failed at a LATER layer: the header/earlier layers read fine, so the file really does end
//--- early => truncated by an interrupted save (Save() is atomic now, so this should be a
//--- legacy file only).
//--- - tiny file: truncated, whatever the layer index.
string likely ;
if ( fileSizeBytes < 65536 )
likely = " the file is only " + IntegerToString ( ( int ) fileSizeBytes ) + " bytes => TRUNCATED (interrupted save, pre-atomic-Save file) " ;
else
if ( failedLayer = = 0 )
likely = " the file is " + IntegerToString ( ( int ) ( fileSizeBytes / 1024 ) ) + " KB and it failed on the FIRST layer, before any layer data mattered "
+ " => this is a READ-PATH CODE fault, not a bad file and not a " + backend + " allocation problem. "
+ " Check that CLayer::CreateElement still exactly matches CArrayObj::CreateElement's signature (const int) so it overrides it, "
+ " and that CLayer::Load/CNeuronBaseOCL::Load are in sync with Save() " ;
else
likely = " the file is " + IntegerToString ( ( int ) ( fileSizeBytes / 1024 ) ) + " KB and layers 0.. " + IntegerToString ( failedLayer - 1 ) +
" read fine => the file ends early (TRUNCATED by an interrupted save) or a neuron tensor could not be allocated on the " + backend ;
Print ( " CNet::Load: " + file_name + " - REJECTED: only loaded " + IntegerToString ( layers . Total ( ) ) +
" of " + IntegerToString ( num ) + " layers (failed at layer " + IntegerToString ( failedLayer ) +
" ). Diagnosis: " + likely + " . " ) ;
}
return ( false ) ;
}
return true ;
}
# endif