Keep name-lookup acceleration unregistered: two failed attempts recorded
sub_4F3704 (resource lookup by name) is 18.8% of load-time samples and does a linear strcmp scan over 6232 entries because the game's own hash cache is disabled (*(self+8) == 0, confirmed by a live probe: 320,000+ calls per load). Attempt 1 crashed the process: guest strings were read through G2H and scanned for a NUL with no bounds check, so a bad offset walked off the end of the mapped region (SIGSEGV, SEGV_ACCERR at a host address). Attempt 2 was memory-safe (uc_mem_read everywhere, length and count caps, every cache hit verified against live guest memory, falls through to the guest on any doubt) but caused a visible frame-rate drop on the prologue loading screen. The cause was a design flaw the earlier probe log had already shown and I misread: the cache was keyed on the table ADDRESS, yet this game reuses one address for different tables (35 and 59 entries alternating in the log). The descriptor comparison therefore marked the index stale on nearly every call, and each rebuild re-read all ~6232 entry strings byte-by-byte - far more work than the scan it replaced. Kept in the tree, unregistered, with both mistakes documented. The fix for a third attempt is to key the cache on the DESCRIPTOR CONTENTS rather than the address, so alternating tables each keep their own index, plus bulk string reads instead of per-byte. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -3116,16 +3116,20 @@ uc_engine* GuestEngine::CreateConfiguredEngine() {
|
|||||||
zlib_accel::kCrc32Addr, zlib_accel::kCrc32Addr);
|
zlib_accel::kCrc32Addr, zlib_accel::kCrc32Addr);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Native resource-name lookup: REGISTRATION DISABLED 2026-09-19 after it
|
// Native resource-name lookup: REGISTRATION DISABLED AGAIN 2026-09-19.
|
||||||
// crashed the process (SIGSEGV, SEGV_ACCERR at a host address, inside the
|
// The safe rewrite no longer crashes, but it made the prologue load
|
||||||
// index build walking a guest string past the end of the mapped region).
|
// visibly LAG - a performance regression, caught live. Cause is in the
|
||||||
// Two wrong assumptions, both visible in the same capture: guest pointers
|
// design, not the safety checks: the cache is keyed on the table
|
||||||
// were dereferenced via G2H and scanned for a NUL without any bounds
|
// ADDRESS, and this game legitimately reuses one address for different
|
||||||
// check, and the table identity assumption was wrong - the log shows one
|
// tables (the probe log showed 35 and 59 entries alternating). So the
|
||||||
// address (0x3537fac4) alternating between 35 and 59 entries, i.e.
|
// descriptor compare marks the index stale on almost every call, and
|
||||||
// different objects living at the same address, so caching keyed on the
|
// each rebuild re-reads all ~6232 entry strings byte-by-byte through
|
||||||
// address alone is invalid. See name_lookup_accel.h for what still needs
|
// uc_mem_read - far more work than the linear strcmp scan it replaces.
|
||||||
// solving; the code is kept, unregistered, as the starting point.
|
// Fix is to key the cache on the DESCRIPTOR CONTENTS (entry arrays,
|
||||||
|
// string pools, threshold, counts) so alternating tables each keep their
|
||||||
|
// own index and nothing rebuilds; plus bulk string reads instead of
|
||||||
|
// per-byte. Kept unregistered until that is done - see
|
||||||
|
// name_lookup_accel.h.
|
||||||
|
|
||||||
// See MeasureAdvanceProbeHookCb's own comment - task #39's root-cause
|
// See MeasureAdvanceProbeHookCb's own comment - task #39's root-cause
|
||||||
// chain, catching the advance value at the exact instruction that
|
// chain, catching the advance value at the exact instruction that
|
||||||
|
|||||||
@@ -22,73 +22,102 @@ constexpr uint32_t kOffPoolA = 224;
|
|||||||
constexpr uint32_t kOffPoolThreshold = 228;
|
constexpr uint32_t kOffPoolThreshold = 228;
|
||||||
constexpr uint32_t kOffPoolB = 232;
|
constexpr uint32_t kOffPoolB = 232;
|
||||||
|
|
||||||
|
// Refuses to index anything implausible. A garbage count field would
|
||||||
|
// otherwise turn the build loop into a multi-million-iteration walk over
|
||||||
|
// arbitrary memory - the first version had no such guard and paid for it.
|
||||||
|
constexpr uint32_t kMaxEntries = 200000;
|
||||||
|
constexpr uint32_t kMaxNameLen = 128;
|
||||||
|
|
||||||
struct TableCache {
|
struct TableCache {
|
||||||
uint32_t countA = 0;
|
uint32_t countA = 0;
|
||||||
uint32_t countB = 0;
|
uint32_t countB = 0;
|
||||||
// name -> entry index. First occurrence wins, matching the guest loop,
|
uint32_t entriesA = 0;
|
||||||
// which returns as soon as it finds a match.
|
uint32_t entriesB = 0;
|
||||||
|
uint32_t poolA = 0;
|
||||||
|
uint32_t poolB = 0;
|
||||||
|
uint32_t threshold = 0;
|
||||||
std::unordered_map<std::string, int32_t> byName;
|
std::unordered_map<std::string, int32_t> byName;
|
||||||
};
|
};
|
||||||
|
|
||||||
std::mutex g_mutex;
|
std::mutex g_mutex;
|
||||||
std::map<uint32_t, TableCache> g_tables; // keyed by the guest table address
|
std::map<uint32_t, TableCache> g_tables;
|
||||||
|
|
||||||
std::atomic<uint64_t> g_hits{0};
|
std::atomic<uint64_t> g_served{0};
|
||||||
|
std::atomic<uint64_t> g_fellThrough{0};
|
||||||
std::atomic<uint64_t> g_builds{0};
|
std::atomic<uint64_t> g_builds{0};
|
||||||
|
|
||||||
uint32_t ReadU32(uc_engine* uc, uint32_t addr) {
|
// Every guest read goes through uc_mem_read, which FAILS on an unmapped
|
||||||
uint32_t v = 0;
|
// address instead of handing back a host pointer to walk off the end of the
|
||||||
uc_mem_read(uc, addr, &v, 4);
|
// region. That is the whole difference from the first attempt, which used
|
||||||
return v;
|
// G2H plus an unbounded NUL scan and segfaulted the process.
|
||||||
|
bool TryReadU32(uc_engine* uc, uint32_t addr, uint32_t* out) {
|
||||||
|
return addr && uc_mem_read(uc, addr, out, 4) == UC_ERR_OK;
|
||||||
}
|
}
|
||||||
|
|
||||||
void ReturnToCaller(uc_engine* uc, uint32_t ret) {
|
// Bounded, fault-tolerant guest string read. Returns false if the string is
|
||||||
uint32_t lr = 0;
|
// unterminated within kMaxNameLen or runs into unmapped memory.
|
||||||
uc_reg_read(uc, UC_ARM_REG_LR, &lr);
|
bool TryReadString(uc_engine* uc, uint32_t addr, std::string* out) {
|
||||||
uc_reg_write(uc, UC_ARM_REG_R0, &ret);
|
if (!addr) return false;
|
||||||
uc_reg_write(uc, UC_ARM_REG_PC, &lr);
|
out->clear();
|
||||||
uc_emu_stop(uc);
|
for (uint32_t i = 0; i < kMaxNameLen; i++) {
|
||||||
|
uint8_t c = 0;
|
||||||
|
if (uc_mem_read(uc, addr + i, &c, 1) != UC_ERR_OK) return false;
|
||||||
|
if (!c) return true;
|
||||||
|
out->push_back((char)c);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Reads a NUL-terminated guest string directly out of the shared region.
|
// Resolves entry index -> the guest address of that entry's name, mirroring
|
||||||
std::string ReadGuestString(GuestEngine& eng, uint32_t addr) {
|
// the guest loop's own addressing (two entry arrays, and a threshold that
|
||||||
if (!addr) return {};
|
// selects which of two string pools an offset belongs to).
|
||||||
const char* p = reinterpret_cast<const char*>(eng.G2H(addr));
|
bool EntryNameAddr(uc_engine* uc, const TableCache& t, uint32_t index, uint32_t* out) {
|
||||||
if (!p) return {};
|
uint32_t entry = (index >= t.countA) ? t.entriesB + (index - t.countA) * 8
|
||||||
return std::string(p);
|
: t.entriesA + index * 8;
|
||||||
|
uint32_t off = 0;
|
||||||
|
if (!TryReadU32(uc, entry, &off)) return false;
|
||||||
|
uint32_t base = t.poolA;
|
||||||
|
if (off >= t.threshold) {
|
||||||
|
off -= t.threshold;
|
||||||
|
base = t.poolB;
|
||||||
|
}
|
||||||
|
if (!base) return false;
|
||||||
|
*out = base + off;
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Builds (or rebuilds) the host-side index for one guest table. Mirrors the
|
// Reads the table's own descriptor fields. Returns false if anything looks
|
||||||
// guest loop's own addressing exactly, including the two-array split and the
|
// unreadable or implausible, in which case this layer stays out of the way.
|
||||||
// threshold that selects which string pool an entry's offset belongs to.
|
bool ReadTableDesc(uc_engine* uc, uint32_t self, TableCache* t) {
|
||||||
void BuildIndex(GuestEngine& eng, uc_engine* uc, uint32_t self, TableCache& cache) {
|
if (!TryReadU32(uc, self + kOffCountA, &t->countA)) return false;
|
||||||
cache.byName.clear();
|
if (!TryReadU32(uc, self + kOffCountB, &t->countB)) return false;
|
||||||
cache.countA = ReadU32(uc, self + kOffCountA);
|
if (!TryReadU32(uc, self + kOffEntriesA, &t->entriesA)) return false;
|
||||||
cache.countB = ReadU32(uc, self + kOffCountB);
|
if (!TryReadU32(uc, self + kOffEntriesB, &t->entriesB)) return false;
|
||||||
const uint32_t entriesA = ReadU32(uc, self + kOffEntriesA);
|
if (!TryReadU32(uc, self + kOffPoolA, &t->poolA)) return false;
|
||||||
const uint32_t entriesB = ReadU32(uc, self + kOffEntriesB);
|
if (!TryReadU32(uc, self + kOffPoolB, &t->poolB)) return false;
|
||||||
const uint32_t poolA = ReadU32(uc, self + kOffPoolA);
|
if (!TryReadU32(uc, self + kOffPoolThreshold, &t->threshold)) return false;
|
||||||
const uint32_t poolB = ReadU32(uc, self + kOffPoolB);
|
uint64_t total = (uint64_t)t->countA + t->countB;
|
||||||
const uint32_t threshold = ReadU32(uc, self + kOffPoolThreshold);
|
return total != 0 && total <= kMaxEntries;
|
||||||
|
}
|
||||||
|
|
||||||
const uint32_t total = cache.countA + cache.countB;
|
bool BuildIndex(uc_engine* uc, uint32_t self, TableCache* t) {
|
||||||
cache.byName.reserve(total * 2);
|
TableCache desc;
|
||||||
|
if (!ReadTableDesc(uc, self, &desc)) return false;
|
||||||
|
desc.byName.reserve((size_t)(desc.countA + desc.countB) * 2);
|
||||||
|
|
||||||
|
const uint32_t total = desc.countA + desc.countB;
|
||||||
for (uint32_t i = 0; i < total; i++) {
|
for (uint32_t i = 0; i < total; i++) {
|
||||||
uint32_t entry = (i >= cache.countA)
|
uint32_t nameAddr = 0;
|
||||||
? entriesB + (i - cache.countA) * 8
|
if (!EntryNameAddr(uc, desc, i, &nameAddr)) return false;
|
||||||
: entriesA + i * 8;
|
std::string name;
|
||||||
uint32_t off = ReadU32(uc, entry);
|
if (!TryReadString(uc, nameAddr, &name)) return false;
|
||||||
uint32_t base = poolA;
|
|
||||||
if (off >= threshold) {
|
|
||||||
off -= threshold;
|
|
||||||
base = poolB;
|
|
||||||
}
|
|
||||||
std::string name = ReadGuestString(eng, base + off);
|
|
||||||
if (name.empty()) continue;
|
if (name.empty()) continue;
|
||||||
cache.byName.emplace(std::move(name), (int32_t)i);
|
// First occurrence wins - the guest loop returns on its first match.
|
||||||
|
desc.byName.emplace(std::move(name), (int32_t)i);
|
||||||
}
|
}
|
||||||
Log("name_lookup_accel: indexed table 0x%x - %u entries (%u+%u), %zu distinct names",
|
*t = std::move(desc);
|
||||||
self, total, cache.countA, cache.countB, cache.byName.size());
|
g_builds.fetch_add(1, std::memory_order_relaxed);
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
@@ -97,42 +126,77 @@ void HookCb(uc_engine* uc, uint64_t, uint32_t, void*) {
|
|||||||
uint32_t self = 0, namePtr = 0;
|
uint32_t self = 0, namePtr = 0;
|
||||||
uc_reg_read(uc, UC_ARM_REG_R0, &self);
|
uc_reg_read(uc, UC_ARM_REG_R0, &self);
|
||||||
uc_reg_read(uc, UC_ARM_REG_R1, &namePtr);
|
uc_reg_read(uc, UC_ARM_REG_R1, &namePtr);
|
||||||
if (!self || !namePtr) return; // let the guest handle its own edge cases
|
if (!self || !namePtr) return;
|
||||||
|
|
||||||
// If the guest's own cache is ever switched on, step aside entirely -
|
// If the guest's own hash cache is ever enabled, step aside - that path
|
||||||
// that path also WRITES into guest structures (it memoises the result),
|
// also WRITES into guest structures (it memoises), and reproducing that
|
||||||
// and duplicating that behaviour here would be guesswork.
|
// here would be guesswork.
|
||||||
uint8_t cacheFlag = 0;
|
uint8_t cacheFlag = 0;
|
||||||
if (uc_mem_read(uc, self + kOffCacheFlag, &cacheFlag, 1) != UC_ERR_OK) return;
|
if (uc_mem_read(uc, self + kOffCacheFlag, &cacheFlag, 1) != UC_ERR_OK) return;
|
||||||
if (cacheFlag) return;
|
if (cacheFlag) return;
|
||||||
|
|
||||||
auto& eng = GuestEngine::Instance();
|
std::string query;
|
||||||
std::string name = ReadGuestString(eng, namePtr);
|
if (!TryReadString(uc, namePtr, &query) || query.empty()) return;
|
||||||
if (name.empty()) return;
|
|
||||||
|
|
||||||
int32_t result = -1;
|
int32_t candidate = -1;
|
||||||
{
|
{
|
||||||
std::lock_guard<std::mutex> lock(g_mutex);
|
std::lock_guard<std::mutex> lock(g_mutex);
|
||||||
TableCache& cache = g_tables[self];
|
TableCache& cache = g_tables[self];
|
||||||
// Rebuild when the table has grown/shrunk since it was indexed. The
|
|
||||||
// counts are the same two fields the guest loop bounds itself with,
|
// The first attempt keyed the cache on the table ADDRESS and treated
|
||||||
// so any change it can see, this sees too.
|
// a changed entry count as "the table grew". The live log disproved
|
||||||
uint32_t countA = ReadU32(uc, self + kOffCountA);
|
// that: one address alternated between 35 and 59 entries, i.e.
|
||||||
uint32_t countB = ReadU32(uc, self + kOffCountB);
|
// DIFFERENT objects reusing the same address. So the descriptor is
|
||||||
if (cache.byName.empty() || cache.countA != countA || cache.countB != countB) {
|
// re-read every call (seven cheap word reads) and the index is
|
||||||
BuildIndex(eng, uc, self, cache);
|
// rebuilt whenever any of it moved.
|
||||||
g_builds.fetch_add(1, std::memory_order_relaxed);
|
TableCache desc;
|
||||||
}
|
if (!ReadTableDesc(uc, self, &desc)) return;
|
||||||
auto it = cache.byName.find(name);
|
const bool stale = cache.byName.empty() || cache.countA != desc.countA ||
|
||||||
if (it != cache.byName.end()) result = it->second;
|
cache.countB != desc.countB || cache.entriesA != desc.entriesA ||
|
||||||
|
cache.entriesB != desc.entriesB || cache.poolA != desc.poolA ||
|
||||||
|
cache.poolB != desc.poolB || cache.threshold != desc.threshold;
|
||||||
|
if (stale && !BuildIndex(uc, self, &cache)) {
|
||||||
|
g_tables.erase(self);
|
||||||
|
return; // could not index safely - let the guest do its own scan
|
||||||
}
|
}
|
||||||
|
|
||||||
uint64_t n = g_hits.fetch_add(1, std::memory_order_relaxed) + 1;
|
auto it = cache.byName.find(query);
|
||||||
if (n % 100000 == 0) {
|
if (it == cache.byName.end()) {
|
||||||
Log("name_lookup_accel: %llu lookups served natively (%llu index builds)",
|
// Not found is NOT answered from cache: a stale index would turn
|
||||||
(unsigned long long)n, (unsigned long long)g_builds.load(std::memory_order_relaxed));
|
// a real entry into a false "absent", and absence is exactly what
|
||||||
|
// the caller acts on. Let the guest's own scan decide.
|
||||||
|
g_fellThrough.fetch_add(1, std::memory_order_relaxed);
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
ReturnToCaller(uc, (uint32_t)result);
|
candidate = it->second;
|
||||||
|
|
||||||
|
// Verify the hit against live guest memory before trusting it. This
|
||||||
|
// is what makes a wrong structural assumption cost performance
|
||||||
|
// instead of correctness.
|
||||||
|
uint32_t nameAddr = 0;
|
||||||
|
std::string actual;
|
||||||
|
if (!EntryNameAddr(uc, cache, (uint32_t)candidate, &nameAddr) ||
|
||||||
|
!TryReadString(uc, nameAddr, &actual) || actual != query) {
|
||||||
|
g_tables.erase(self);
|
||||||
|
g_fellThrough.fetch_add(1, std::memory_order_relaxed);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t n = g_served.fetch_add(1, std::memory_order_relaxed) + 1;
|
||||||
|
if (n % 200000 == 0) {
|
||||||
|
Log("name_lookup_accel: %llu lookups served natively, %llu fell through to the guest, "
|
||||||
|
"%llu index builds",
|
||||||
|
(unsigned long long)n,
|
||||||
|
(unsigned long long)g_fellThrough.load(std::memory_order_relaxed),
|
||||||
|
(unsigned long long)g_builds.load(std::memory_order_relaxed));
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t lr = 0, ret = (uint32_t)candidate;
|
||||||
|
uc_reg_read(uc, UC_ARM_REG_LR, &lr);
|
||||||
|
uc_reg_write(uc, UC_ARM_REG_R0, &ret);
|
||||||
|
uc_reg_write(uc, UC_ARM_REG_PC, &lr);
|
||||||
|
uc_emu_stop(uc);
|
||||||
}
|
}
|
||||||
|
|
||||||
} // namespace name_lookup_accel
|
} // namespace name_lookup_accel
|
||||||
|
|||||||
Reference in New Issue
Block a user