(编辑:我在这个问题中添加了一个new answer,它使挂墙时间减少了 95%。)
我制作了一个最小工作示例来说明您要解决的问题。这是您在问题中应该始终做的事情。
然后我删除了unsigned long long int 的东西,并用cstdint 库中的uint64_t 替换它。这确保我们在相同的数据大小上运行,因为 unsigned long long int 可能意味着几乎任何东西,具体取决于您使用的计算机/编译器。
生成的 MWE 如下所示:
#include <chrono>
#include <cstdint>
#include <cstdio>
#include <deque>
#include <functional>
#include <iostream>
#include <random>
#include <unordered_map>
#include <vector>
typedef std::unordered_map<uint64_t, char> table_t;
const int TEST_TABLE_SIZE = 10000000;
void Save(const table_t &map){
std::cout<<"Save. ";
const auto start = std::chrono::steady_clock::now();
FILE *f = fopen("/z/map", "wb");
for(auto iter=map.begin(); iter!=map.end(); iter++){
fwrite(&(iter->first), 8, 1, f);
fwrite(&(iter->second), 1, 1, f);
}
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Save time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
//Take advantage of the limited range of values to save time
void SaveLookup(const table_t &map){
std::cout<<"SaveLookup. ";
const auto start = std::chrono::steady_clock::now();
//Create a lookup table
std::vector< std::deque<uint64_t> > lookup(256);
for(auto &kv: map)
lookup.at(kv.second+128).emplace_back(kv.first);
//Save lookup table header
FILE *f = fopen("/z/map", "wb");
for(const auto &row: lookup){
const uint32_t rowsize = row.size();
fwrite(&rowsize, 4, 1, f);
}
//Save values
for(const auto &row: lookup)
for(const auto &val: row)
fwrite(&val, 8, 1, f);
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Save time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
//Take advantage of the limited range of values and contiguous memory to
//save time
void SaveLookupVector(const table_t &map){
std::cout<<"SaveLookupVector. ";
const auto start = std::chrono::steady_clock::now();
//Create a lookup table
std::vector< std::vector<uint64_t> > lookup(256);
for(auto &kv: map)
lookup.at(kv.second+128).emplace_back(kv.first);
//Save lookup table header
FILE *f = fopen("/z/map", "wb");
for(const auto &row: lookup){
const uint32_t rowsize = row.size();
fwrite(&rowsize, 4, 1, f);
}
//Save values
for(const auto &row: lookup)
fwrite(row.data(), 8, row.size(), f);
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Save time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
void Load(table_t &map){
std::cout<<"Load. ";
const auto start = std::chrono::steady_clock::now();
FILE *f = fopen("/z/map", "rb");
uint64_t key;
char val;
while(fread(&key, 8, 1, f)){
fread(&val, 1, 1, f);
map[key] = val;
}
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Load time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
void Load2(table_t &map){
std::cout<<"Load with Reserve. ";
map.reserve(TEST_TABLE_SIZE+TEST_TABLE_SIZE/8);
const auto start = std::chrono::steady_clock::now();
FILE *f = fopen("/z/map", "rb");
uint64_t key;
char val;
while(fread(&key, 8, 1, f)){
fread(&val, 1, 1, f);
map[key] = val;
}
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Load time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
//Take advantage of the limited range of values to save time
void LoadLookup(table_t &map){
std::cout<<"LoadLookup. ";
map.reserve(TEST_TABLE_SIZE+TEST_TABLE_SIZE/8);
const auto start = std::chrono::steady_clock::now();
FILE *f = fopen("/z/map", "rb");
//Read the header
std::vector<uint32_t> inpsizes(256);
for(int i=0;i<256;i++)
fread(&inpsizes[i], 4, 1, f);
uint64_t key;
for(int i=0;i<256;i++){
const char val = i-128;
for(int v=0;v<inpsizes.at(i);v++){
fread(&key, 8, 1, f);
map[key] = val;
}
}
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Load time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
//Take advantage of the limited range of values and contiguous memory to save time
void LoadLookupVector(table_t &map){
std::cout<<"LoadLookupVector. ";
map.reserve(TEST_TABLE_SIZE+TEST_TABLE_SIZE/8);
const auto start = std::chrono::steady_clock::now();
FILE *f = fopen("/z/map", "rb");
//Read the header
std::vector<uint32_t> inpsizes(256);
for(int i=0;i<256;i++)
fread(&inpsizes[i], 4, 1, f);
for(int i=0;i<256;i++){
const char val = i-128;
std::vector<uint64_t> keys(inpsizes[i]);
fread(keys.data(), 8, inpsizes[i], f);
for(const auto &key: keys)
map[key] = val;
}
fclose(f);
const auto end = std::chrono::steady_clock::now();
std::cout<<"Load time = "<< std::chrono::duration<double, std::milli> (end-start).count() << " ms" << std::endl;
}
int main(){
//Perfectly horrendous way of seeding a PRNG, but we'll do it here for brevity
auto generator = std::mt19937(12345); //Combination of my luggage
//Generate values within the specified closed intervals
auto key_rand = std::bind(std::uniform_int_distribution<uint64_t>(0,std::numeric_limits<uint64_t>::max()), generator);
auto val_rand = std::bind(std::uniform_int_distribution<int>(std::numeric_limits<char>::lowest(),std::numeric_limits<char>::max()), generator);
std::cout<<"Generating test data..."<<std::endl;
//Generate a test table
table_t map;
for(int i=0;i<TEST_TABLE_SIZE;i++)
map[key_rand()] = (char)val_rand(); //Low chance of collisions, so we get quite close to the desired size
Save(map);
{ table_t map2; Load (map2); }
{ table_t map2; Load2(map2); }
SaveLookup(map);
SaveLookupVector(map);
{ table_t map2; LoadLookup (map2); }
{ table_t map2; LoadLookupVector(map2); }
}
在我使用的测试数据集上,这给了我 1982 毫秒的写入时间和 7467 毫秒的读取时间(使用您的原始代码)。似乎读取时间是最大的瓶颈,所以我创建了一个新函数Load2,它在读取之前为 unordered_map 保留足够的空间。这将读取时间减少到 4700 毫秒(节省 37%)。
编辑 1
现在,我注意到您的 unordered_map 的值只能采用 255 个不同的值。因此,我可以轻松地将unordered_map 转换为RAM 中的一种查找表。也就是说,而不是:
123123 1
234234 0
345345 1
237872 1
我可以重新排列数据,使其看起来像:
0 234234
1 123123 345345 237872
这样做有什么好处?这意味着我不再需要将值写入磁盘。这样可以为每个表条目节省 1 个字节。由于每个表条目由 8 个字节的键和 1 个字节的值组成,这应该可以节省 11% 的读取和写入时间减去重新排列内存的成本(我希望它很低,因为 RAM) .
最后,一旦我完成了上述重新排列,如果我的机器上有很多空闲 RAM,我可以将所有内容打包到一个向量中并将连续数据读/写到磁盘。
这样做会产生以下时间:
Save. Save time = 1836.52 ms
Load. Load time = 7114.93 ms
Load with Reserve. Load time = 4277.58 ms
SaveLookup. Save time = 1688.73 ms
SaveLookupVector. Save time = 1394.95 ms
LoadLookup. Load time = 3927.3 ms
LoadLookupVector. Load time = 3739.37 ms
请注意,从Save 到SaveLookup 的转换提供了8% 的加速,从Load with Reserve 到LoadLookup 的转换也提供了8% 的加速。这符合我们的理论!
同时使用连续内存可使原始保存时间总共加快 24%,与原始加载时间相比总共加快 47%。