You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

MapViewOfFile内存映射文件性能异常问题求助

大内存映射文件多线程读取性能指数级下降问题求助

我用10个线程读取同一内存映射文件的不同分块,文件较小时(2GB)处理速度极快,但增大到12GB(2000多万条DBTransaction)时,总耗时飙升至26秒,远超预期,寻求原因分析和解决建议。

核心处理代码

testTimer.Start();

u64 startPositionInBytes = startPosition * sizeof(DBTransaction);
u64 chunkSizeInBytes = chunkSize * sizeof(DBTransaction);

u64 viewOffset = (startPositionInBytes / allocGranularity) * allocGranularity;

u64 mapViewSize = (startPositionInBytes % allocGranularity) + chunkSizeInBytes;
if (mapViewSize > (queryThread->FileSize - viewOffset))
    mapViewSize = (queryThread->FileSize - viewOffset);

u64 dataReadOffset = startPositionInBytes - viewOffset;

DWORD high = (DWORD)(viewOffset >> 32);
DWORD low = (DWORD)(viewOffset & 0xFFFFFFFF);
u8* fileData = (u8*)MapViewOfFile(queryThread->FileMap, FILE_MAP_READ, high, low, mapViewSize);
if (fileData == nullptr)
{
    RALogError("Unable to create a file view into mapped file at position %llu (chunk size %llu). Error was: %s. Application can not proceed", startPosition, chunkSize, GetLastSystemError().c_str());
}
elapsed = testTimer.Measure();
queryThread->MapGlobalTime += elapsed;

DBTransaction* transaction = (DBTransaction*)(fileData + dataReadOffset);

for (u32 i = 0; i < chunkSize; ++i)
{
    testTimer.Start();
    //DBTransaction* transaction = (DBTransaction*)pBytes;

    queryThread->TotalRead++;

    elapsed = testTimer.Measure();
    queryThread->CopyTime += elapsed;

    testTimer.Start();
    
    bool allConditionsPassed = true;

    for (u32 j = 0; j < QueryFilters.size(); ++j)
    {
        RAQueryFilter* filter = &QueryFilters[j];
        bool atLeastOneTrue = false;
        for (u32 k = 0; k < filter->Conditions.size(); ++k)
        {
            QueryCondition* condition = &filter->Conditions[k];

            if (condition->Compare(transaction, condition))
                atLeastOneTrue = true;

            if (atLeastOneTrue)
                break;
        }

        if (atLeastOneTrue == false)
        {
            allConditionsPassed = false;
            break;
        }
    }
    elapsed = testTimer.Measure();
    queryThread->MatchGlobalTime += elapsed;
    
    testTimer.Start();
    if (allConditionsPassed)
    {
        queryPointer->AddToResult(queryThread, queryPointer, transaction);
        queryThread->TotalMatches++;
    }
    elapsed = testTimer.Measure();
    queryThread->ComputeGlobalTime += elapsed;

    transaction ++;
}

线程创建代码

u32 requiredThreads = AvailableThreadsCount;
u64 transactionsPerThread = (u64)ceil((f32)dataToRead / (f32)requiredThreads);

// Make sure to not make thread load too small. If the number of transactions is less than 500K, reduce the number of used threads
if (transactionsPerThread < RA_MINIMUM_THREAD_TRANSACTION_COUNT)
{
    requiredThreads = (u32)ceil((f32)dataToRead / (f32)RA_MINIMUM_THREAD_TRANSACTION_COUNT);
    transactionsPerThread = RA_MINIMUM_THREAD_TRANSACTION_COUNT;
}

// Spawn the threads
u32 i = 0;
while (dataToRead > 0)
{
    u64 chunkSize = dataToRead;
    if (chunkSize > transactionsPerThread)
        chunkSize = transactionsPerThread;

    RA_ASSERT(i < requiredThreads, "Too many threads created!");
    sQueryThread* queryThread = &QueryThread[i];

#ifdef USE_FILE_MAPPING
    queryThread->Set(startPosition, chunkSize, testReadChunk, TransactionsHistory->FileSizeInBytes, TransactionsHistory->ReadFileMapHandle);
#else 
    queryThread->Set(startPosition, chunkSize, 0, NULL);
#endif

    //RALog("Spawning thread %i", i);
    queryThread->Handle = CreateThread(NULL, 0, (LPTHREAD_START_ROUTINE)DoLinearSearchQueryThread, queryThread, 0, &queryThread->ThreadID);

    i++;
    startPosition += chunkSize;
    dataToRead -= chunkSize;
}

性能分析结论

  • MapViewOfFile耗时仅数毫秒,并非性能瓶颈
  • 最初耗时集中在Compare函数,注释该函数后耗时转移到AddToResult;再注释AddToResult,仅复制DBTransaction时耗时依旧很高,推测真正的耗时点是首次解引用DBTransaction指针

原因分析

  1. TLB失效:12GB文件远超CPU TLB(地址转换旁路缓冲)的覆盖范围,多线程访问分散的分块会导致大量TLB miss。每次解引用指针都需要触发页表遍历,甚至从磁盘加载页(内存不足时),延迟随文件大小呈指数级上升。
  2. 缓存颠簸与页置换:当物理内存不足以容纳所有映射页时,操作系统会频繁进行页置换,多线程的并发访问会把其他线程的缓存页挤出去,重复的页加载操作大幅增加总耗时。
  3. 分块对齐问题:代码中viewOffset按allocGranularity对齐,但线程分块起始位置可能跨大页边界,进一步加剧TLB失效和缓存竞争。

解决建议

  • 启用大页内存:
    • Windows下通过CreateFileMapping指定SEC_LARGE_PAGES标志(需提前配置权限),将文件映射到2MB/1GB大页,大幅减少TLB miss次数。
    • 确保线程分块起始位置严格对齐到大页边界,避免跨页访问。
  • 调整分块与线程数量:
    • 增大单个线程处理的分块大小(参考CPU L3缓存容量),减少线程数量,提升缓存利用率,降低缓存颠簸。
    • 当文件大小超过物理内存70%时,进一步减少线程数,避免过度页置换。
  • 预加载与内存锁定:
    • 使用PrefetchVirtualMemory提前加载需要访问的文件页到物理内存,避免运行时页加载延迟。
    • 对高频访问页使用VirtualLock锁定到物理内存(需足够物理内存支持),防止被置换。
  • 优化数据访问模式:
    • 保持顺序访问模式(当前已实现),触发CPU预取机制,提前加载后续数据到缓存。
    • 若DBTransaction结构较大,可将过滤用到的字段单独存储在文件头部,减少每次访问的数据量。

内容的提问来源于stack exchange,提问作者Pierluigi Serra

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 19:05:20