Given the following benchmark:
const char* payload = "abcdefghijk";
const std::size_t payload_len = 11;
const std::size_t payload_count = 1000;
static void StringAppend(benchmark::State& state) {
for (auto _ : state) {
std::string created_string;
created_string.reserve(payload_len * payload_count + 1);
for(int i = 0 ; i < payload_count; ++i) {
created_string.append(payload, payload_len);
}
benchmark::DoNotOptimize(created_string);
}
}
BENCHMARK(StringAppend);
static void StringBackInsert(benchmark::State& state) {
for (auto _ : state) {
std::string created_string;
created_string.reserve(payload_len * payload_count + 1);
auto inserter = std::back_inserter(created_string);
for(int i = 0 ; i < payload_count; ++i) {
for(std::size_t i = 0; i < payload_len; ++i) {
*inserter = payload[i];
++inserter;
}
}
benchmark::DoNotOptimize(created_string);
}
}
BENCHMARK(StringBackInsert);
static void StringPushBack(benchmark::State& state) {
for (auto _ : state) {
std::string created_string;
created_string.reserve(payload_len * payload_count + 1);
for(int i = 0 ; i < payload_count; ++i) {
for(std::size_t i = 0; i < payload_len; ++i) {
created_string.push_back(payload[i]);
}
}
benchmark::DoNotOptimize(created_string);
}
}
BENCHMARK(StringPushBack);
I get the following on quickbench, which show a very dramatic difference:

Considering that all the required memory is allocated ahead of time, I'm having a lot of trouble buying into the idea that just doing the size vs capacity check represents essentially all of the cost here, unless maybe there's a massive number of load-hit-store or branch misprediction involved.
http://quick-bench.com/XQ9kepYFE1_dZD8vVaQwOUSSVoE
What I'd like to understand is:
- Is there something specific to this setup that the compiler is using that will not necessarily apply in a real-world scenario?
- If so, is there a way to rearrange this benchmark to be more representative?