diff --git a/lib/tdeck_ui/UI/LXMF/NomadNetCache.cpp b/lib/tdeck_ui/UI/LXMF/NomadNetCache.cpp index 76de69c5..3826c8d3 100644 --- a/lib/tdeck_ui/UI/LXMF/NomadNetCache.cpp +++ b/lib/tdeck_ui/UI/LXMF/NomadNetCache.cpp @@ -929,6 +929,38 @@ CacheResult NomadNetCache::invalidate(const CacheKey& key) { } void NomadNetCache::service() { + // Transient-stall guard (cross-call): compare this call's entry state to + // the previous call's. A transient (BUSY/UNAVAILABLE) retry changes none + // of the tracked bits, so an unchanged entry across consecutive service() + // calls is a stall; any advance resets the counter. Past the bounded + // budget, bail so a persistently unhealthy SD seam can never pin the cache + // (and the NomadNet UI at "Checking SD page cache..."). The baseline is + // refreshed before the switch so every path — including early returns — + // leaves a correct entry for the next call. + if (operation_ != Operation::NONE && + operation_ == transient_prev_op_ && + offset_ == transient_prev_offset_ && + scan_index_ == transient_prev_scan_index_ && + cleanup_index_ == transient_prev_cleanup_index_ && + scan_seen_ == transient_prev_scan_seen_ && + read_open_ == transient_prev_read_open_ && + write_open_ == transient_prev_write_open_) { + if (transient_stall_count_ < MAX_TRANSIENT_STALL_TICKS) { + ++transient_stall_count_; + } else { + transient_stall_count_ = 0; + transient_bail(); + } + } else if (operation_ != Operation::NONE) { + transient_stall_count_ = 0; + } + transient_prev_op_ = operation_; + transient_prev_offset_ = offset_; + transient_prev_scan_index_ = scan_index_; + transient_prev_cleanup_index_ = cleanup_index_; + transient_prev_scan_seen_ = scan_seen_; + transient_prev_read_open_ = read_open_; + transient_prev_write_open_ = write_open_; switch (operation_) { case Operation::NONE: return; @@ -1428,6 +1460,30 @@ void NomadNetCache::service() { } } +void NomadNetCache::transient_bail() { + // The storage seam has been stalled for far longer than any real SPI + // contention or SD mount window. Stop retrying: drop all namespace + // authority (lookups and commits now bypass), and clear the in-flight op + // so the caller's flow falls through to a live fetch, restoring the + // pre-cache page-load behavior instead of a frozen UI. + transient_stall_count_ = 0; + namespace_authoritative_ = false; + read_open_ = false; + write_open_ = false; + commit_job_ = false; + cleanup_stages_after_failure_ = false; + eviction_pending_ = false; + eviction_generation_ = -1; + quota_recovery_ = false; + recovery_complete_ = false; + offset_ = 0; + io_.clear(); + metadata_bytes_.clear(); + ExternalVector().swap(body_); + operation_ = Operation::NONE; + result_ = CacheResult::BYPASS; +} + void NomadNetCache::cancel() { if (!busy() || eviction_pending_ || quota_recovery_) { return; diff --git a/lib/tdeck_ui/UI/LXMF/NomadNetCache.h b/lib/tdeck_ui/UI/LXMF/NomadNetCache.h index 771267d6..c3d7f46c 100644 --- a/lib/tdeck_ui/UI/LXMF/NomadNetCache.h +++ b/lib/tdeck_ui/UI/LXMF/NomadNetCache.h @@ -226,6 +226,30 @@ private: CacheResult quota_result_ = CacheResult::STORED; bool quota_recovery_ = false; + // Transient-stall guard. A storage op that returns BUSY/UNAVAILABLE (SPI + // mutex starved, card not mounted) is retried forever by the step machine; + // on a persistently unhealthy seam that pins operation_ != NONE and the + // NomadNet UI freezes at "Checking SD page cache...". service() compares + // this call's entry state to the previous call's: any advance (op change, + // offset, scan/cleanup index, scan count, open-flag) resets the stall + // counter, so slow-but-progressing steps (chunked reads/writes, directory + // scans) never false-trip; a tick with no advance is a transient stall. + // Past the bounded budget the cache bails: mark the namespace + // non-authoritative for the session (lookups/commits then bypass) and + // clear the op, so the flow falls through to a live fetch (the pre-cache + // behavior). 500 no-progress ticks far exceeds any real SPI contention or + // SD mount window (each op already waits only 100 ms on the bus mutex). + static constexpr std::uint32_t MAX_TRANSIENT_STALL_TICKS = 500; + Operation transient_prev_op_ = Operation::NONE; + std::size_t transient_prev_offset_ = 0; + std::size_t transient_prev_scan_index_ = 0; + std::size_t transient_prev_cleanup_index_ = 0; + std::size_t transient_prev_scan_seen_ = 0; + bool transient_prev_read_open_ = false; + bool transient_prev_write_open_ = false; + std::uint32_t transient_stall_count_ = 0; + void transient_bail(); + static constexpr std::size_t VERIFY_SCRATCH_BYTES = 1024; std::array verify_scratch_{}; std::uint64_t verify_hash_ = 1469598103934665603ULL; diff --git a/tests/native/test_nomadnet_cache.cpp b/tests/native/test_nomadnet_cache.cpp index a1b353bd..24e58729 100644 --- a/tests/native/test_nomadnet_cache.cpp +++ b/tests/native/test_nomadnet_cache.cpp @@ -26,6 +26,25 @@ struct MemoryStorage final:NomadNetStorage{ StorageResult endList()override{++operations;return StorageResult::OK;}std::vectorlist;size_t li=0; }; static CacheKey key(const char*path="/page/index.mu"){return CacheKey{"0123456789abcdef0123456789abcdef",path,RequestDataClass::NIL};} +// A seam whose directory enumeration is permanently BUSY (SPI mutex starved +// for longer than any real contention window) — the exact condition that +// pinned the boot-time recovery and froze the UI at "Checking SD page cache...". +struct HangListStorage final:NomadNetStorage{ + StorageResult hang = StorageResult::OK; + bool isAvailable()const override{return true;} + StorageResult beginRead(const char*,uint32_t&s)override{s=0;return StorageResult::MISS;} + StorageResult readChunk(uint8_t*,size_t,size_t&n)override{n=0;return StorageResult::OK;} + StorageResult endRead()override{return StorageResult::OK;} + StorageResult beginWrite(const char*)override{return StorageResult::OK;} + StorageResult writeChunk(const uint8_t*,size_t z,size_t&n)override{n=z;return StorageResult::OK;} + StorageResult commitWrite()override{return StorageResult::OK;} + StorageResult abortWrite()override{return StorageResult::OK;} + StorageResult remove(const char*)override{return StorageResult::MISS;} + StorageResult rename(const char*,const char*)override{return StorageResult::MISS;} + StorageResult stat(const char*,uint32_t&)override{return StorageResult::MISS;} + StorageResult beginList(const char*)override{return hang;} + StorageResult nextList(char*,size_t,bool&d)override{d=true;return StorageResult::OK;} + StorageResult endList()override{return StorageResult::OK;}}; static void drain(NomadNetCache&c,MemoryStorage*s=nullptr){for(int i=0;i<1000&&c.busy();++i){const auto before=s?s->operations:0;c.service();if(s&&s->operations-before>1)throw std::runtime_error("more than one storage operation per service tick");}} int main(){int f=0;auto ck=[&](bool x,const char*n){if(!x){++f;std::cerr<<"FAIL "< body={'h','e','l','l','o'}; ck(canonical_cache_key(key())=="0123456789abcdef0123456789abcdef\n/page/index.mu\nnil","canonical key"); @@ -45,7 +64,16 @@ int main(){int f=0;auto ck=[&](bool x,const char*n){if(!x){++f;std::cerr<<"FAIL // Interrupted promotion keeps a valid prior generation. MemoryStorage s2;NomadNetCache c2(s2,cfg);drain(c2,&s2);ck(c2.beginCommit(key(),body,300,60)==CacheResult::PENDING,"seed");drain(c2);s2.fail_rename_at=2;ck(c2.beginCommit(key(),std::vector{'x'},301,60)==CacheResult::PENDING,"crash commit");drain(c2);ck(c2.lastResult()==CacheResult::STORAGE_ERROR,"rename crash reported");s2.fail_rename_at=0;ck(c2.beginLookup(key(),302)==CacheResult::PENDING,"fallback after crash");drain(c2);ck(c2.takeBody(got)&&std::equal(got.begin(),got.end(),body.begin(),body.end()),"prior generation survives"); // SD faults are cache misses to caller, never page failures. - s2.available=false;ck(c2.beginLookup(key(),302)==CacheResult::PENDING,"unavailable begins");drain(c2);ck(c2.lastResult()==CacheResult::BYPASS,"unavailable bypass");s2.available=true;s2.busy=true;ck(c2.beginLookup(key(),302)==CacheResult::PENDING,"busy begins");drain(c2);ck(c2.lastResult()==CacheResult::BYPASS,"busy bypass"); + s2.available=false;ck(c2.beginLookup(key(),302)==CacheResult::PENDING,"unavailable begins");drain(c2);ck(c2.lastResult()==CacheResult::BYPASS,"unavailable bypass");s2.available=true;s2.busy=true;ck(c2.beginLookup(key(),302)==CacheResult::PENDING,"busy begins");drain(c2);ck(c2.lastResult()==CacheResult::BYPASS,"busy bypass");s2.busy=false; + // A persistently transient seam during boot-time recovery must not pin the + // cache (and the UI). Before this fix the recovery retried beginList forever; + // now the stall budget expires, the namespace is marked non-authoritative, + // and the op clears so the flow can fall through to a live fetch. + MemoryStorage busy_boot;busy_boot.available=false;NomadNetCache bc(busy_boot,cfg); + for(int i=0;i<2000&&bc.busy();++i)bc.service(); + ck(!bc.busy(),"persistent transient recovery bails instead of pinning"); + ck(!bc.recoveryComplete(),"bailed recovery is not authoritative"); + ck(bc.beginLookup(key(),100)==CacheResult::BYPASS,"post-bail lookup bypasses to live"); // Deterministic expired-first then oldest quota eviction. MemoryStorage s3;NomadNetCache c3(s3,cfg);drain(c3,&s3);for(int i=0;i<3;i++){auto k=key((std::string("/page/")+char('a'+i)).c_str());ck(c3.beginCommit(k,body,400+i,i==0?1:100)==CacheResult::PENDING,"quota commit");drain(c3);}ck(c3.entryCount()<=2&&c3.totalBytes()<=cfg.max_bytes,"quotas bounded");ck(c3.beginLookup(key("/page/a"),500)==CacheResult::PENDING,"evicted lookup");drain(c3);ck(c3.lastResult()!=CacheResult::HIT,"expired evicted first"); diff --git a/tests/native/test_nomadnet_cache_flow.cpp b/tests/native/test_nomadnet_cache_flow.cpp index c802ad0d..27d4d92f 100644 --- a/tests/native/test_nomadnet_cache_flow.cpp +++ b/tests/native/test_nomadnet_cache_flow.cpp @@ -2,9 +2,10 @@ #include #include #include +#include "NomadNetCache.h" #include "NomadNetCacheFlow.h" using namespace UI::LXMF::NomadNet; -struct Mem:NomadNetStorage{std::map>f;std::string a;size_t p=0;bool w=false;bool isAvailable()const override{return true;}StorageResult beginRead(const char*n,uint32_t&s)override{auto i=f.find(n);if(i==f.end())return StorageResult::MISS;a=n;p=0;s=i->second.size();return StorageResult::OK;}StorageResult readChunk(uint8_t*o,size_t c,size_t&n)override{auto&v=f[a];n=std::min(c,v.size()-p);memcpy(o,v.data()+p,n);p+=n;return StorageResult::OK;}StorageResult endRead()override{return StorageResult::OK;}StorageResult beginWrite(const char*n)override{a=n;f[a].clear();w=true;return StorageResult::OK;}StorageResult writeChunk(const uint8_t*d,size_t z,size_t&n)override{n=z;f[a].insert(f[a].end(),d,d+z);return StorageResult::OK;}StorageResult commitWrite()override{w=false;return StorageResult::OK;}StorageResult abortWrite()override{w=false;return StorageResult::OK;}StorageResult remove(const char*n)override{return f.erase(n)?StorageResult::OK:StorageResult::MISS;}StorageResult rename(const char*x,const char*y)override{auto i=f.find(x);if(i==f.end())return StorageResult::MISS;f[y]=i->second;f.erase(i);return StorageResult::OK;}StorageResult stat(const char*,uint32_t&)override{return StorageResult::MISS;}StorageResult beginList(const char*)override{return StorageResult::OK;}StorageResult nextList(char*,size_t,bool&d)override{d=true;return StorageResult::OK;}StorageResult endList()override{return StorageResult::OK;}}; +struct Mem:NomadNetStorage{std::map>f;std::string a;size_t p=0;bool w=false;bool list_busy=false;bool isAvailable()const override{return true;}StorageResult beginRead(const char*n,uint32_t&s)override{auto i=f.find(n);if(i==f.end())return StorageResult::MISS;a=n;p=0;s=i->second.size();return StorageResult::OK;}StorageResult readChunk(uint8_t*o,size_t c,size_t&n)override{auto&v=f[a];n=std::min(c,v.size()-p);memcpy(o,v.data()+p,n);p+=n;return StorageResult::OK;}StorageResult endRead()override{return StorageResult::OK;}StorageResult beginWrite(const char*n)override{a=n;f[a].clear();w=true;return StorageResult::OK;}StorageResult writeChunk(const uint8_t*d,size_t z,size_t&n)override{n=z;f[a].insert(f[a].end(),d,d+z);return StorageResult::OK;}StorageResult commitWrite()override{w=false;return StorageResult::OK;}StorageResult abortWrite()override{w=false;return StorageResult::OK;}StorageResult remove(const char*n)override{return f.erase(n)?StorageResult::OK:StorageResult::MISS;}StorageResult rename(const char*x,const char*y)override{auto i=f.find(x);if(i==f.end())return StorageResult::MISS;f[y]=i->second;f.erase(i);return StorageResult::OK;}StorageResult stat(const char*,uint32_t&)override{return StorageResult::MISS;}StorageResult beginList(const char*)override{return list_busy?StorageResult::BUSY:StorageResult::OK;}StorageResult nextList(char*,size_t,bool&d)override{d=true;return StorageResult::OK;}StorageResult endList()override{return StorageResult::OK;}}; int main(){int f=0;auto ck=[&](bool x,const char*n){if(!x){f++;std::cerr<<"FAIL "<b={'o','k'};CacheEligibility e{true,true,false,false,false,false,RequestDataClass::NIL};ck(flow.acceptLive(b,e,100),"valid live accepted");ck(flow.pageReady()&&flow.status()=="Page loaded (live)","render ready before commit");for(int i=0;i<20;i++)flow.service(); NomadNetCacheFlow hit(c);ck(hit.begin(k,101,false)==CacheFlowState::LOOKUP,"second lookup");for(int i=0;i<10&&hit.state()==CacheFlowState::LOOKUP;i++)hit.service();ExternalVectorout;ck(hit.state()==CacheFlowState::READY&&hit.takePage(out)&&std::equal(out.begin(),out.end(),b.begin(),b.end())&&hit.status()=="Cached page; current reachability not checked","hit without peer and without internal-vector copy"); @@ -27,4 +28,17 @@ int main(){int f=0;auto ck=[&](bool x,const char*n){if(!x){f++;std::cerr<<"FAIL ck(recovering_reload.state()==CacheFlowState::NEED_LIVE,"reload starts exactly one live fetch only after terminal invalidation"); ck(recovering.beginLookup(k,102)==CacheResult::PENDING,"post reload invalidation lookup");for(int i=0;i<50&&recovering.busy();++i)recovering.service(); ck(recovering.lastResult()!=CacheResult::HIT,"reload during recovery removed stale generation"); + // The device symptom: boot-time recovery pinned by a permanently BUSY SD + // seam (SPI mutex starved). Before the fix the flow sat in LOOKUP forever + // and the UI froze at "Checking SD page cache..."; now the stall budget + // expires and the flow falls through to a live fetch. + { + Mem hang;hang.list_busy=true;NomadNetCache hc(hang);NomadNetCacheFlow hf(hc); + CacheKey hk{"fedcba9876543210fedcba9876543210","/page/hang.mu",RequestDataClass::NIL}; + ck(hf.begin(hk,1000,false)==CacheFlowState::LOOKUP,"hang lookup admitted"); + int serviced=0; + while(hf.state()==CacheFlowState::LOOKUP&&serviced<10000){hf.service();++serviced;} + ck(hf.state()==CacheFlowState::NEED_LIVE,"pinned recovery no longer freezes the lookup"); + ck(serviced<10000,"lookup reached live in bounded service ticks"); + } std::cout<<(f?"failed":"passed")<<"\n";return f?1:0;}