We're currently facing a somewhat complex problem with mmap performance on our Linux server.
We use a server with 64-core AMD Opteron 6374 and 128GiB of RAM. Here, we created a qemu virtual machine with the same core count and 64GiB of RAM. We use it for unit-testing a program I wrote. There are around 60 unit tests that run in parallel, each of which allocates a little over 1GiB of RAM. Because the process memory compresses really well, we decided to enable Zram. During our tests, the memory usage dropped to around 300MiB for each process, which is a significant gain, at a relatively small performance loss (the swap area stays in physical memory).
Currently, with our tests, we don't swap just yet, but we've observed very poor mmap performance. A single call to mmap, from our testing, could take up to 7 minutes (without swapping, of course; allocating maybe somewhere between 2MBps-20MBps of memory). Sometimes, though all mmaps on all the 60 processes are nearly instant and the processes allocate the required gigabyte of RAM. We watch them allocating tiny amounts of memory per second in real time, though:
The program I wrote follows:
// CC0, inspired by dzaima's code, which was inspired by my code.
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <stdint.h>
#include <sys/mman.h>
#include <signal.h>
#include <unistd.h>
#define u8 uint8_t
#define i32 int32_t
#define u32 uint32_t
#define i64 int64_t
#define u64 uint64_t
#define C const
#define P static
#define _(a...) {return({a;});}
#define F_(n,a...) for(int i=0;i<n;i++){a;}
#define F1(n,x,a...) for(i32 i=0;i<n;i+=x){a;}
#define INLINE P inline __attribute__((always_inline))
#define assert(X) if(!(X))__builtin_unreachable();
#define LKL(c) __builtin_expect((c),1)
typedef u32 W;
#define SZ 19
#define END 1162261467ULL
P C u8 crz[]={1,0,0,9,1,0,2,9,2,2,1},crz2[]={4,3,3,1,0,0,1,0,0,9,9,9,9,9,9,9,4,3,5,1,0,2,1,0,2,9,9,9,9,9,9,9,5,5,4,2,2,1,2,2,1,9,9,9,9,9,9,9,4,3,3,1,0,0,7,6,6,9,9,9,9,9,9,9,4,3,5,1,0,2,7,6,8,9,9,9,9,9,9,9,5,5,4,2,2,1,8,8,7,9,9,9,9,9,9,9,7,6,6,7,6,6,4,3,3,9,9,9,9,9,9,9,7,6,8,7,6,8,4,3,5,9,9,9,9,9,9,9,8,8,7,8,8,7,5,5,4,9,9,9,9,9,9,9};
#define UNR_CRZ(trans,sf1,sf2)W am=a%sf1,ad=a/sf1,dm=d%sf1,dd=d/sf1;r+=k*trans[am+sf2*dm];a=ad;d=dd;k*=sf1;
INLINE W mcrz(W a, W d){W r=0,k=1;
#pragma GCC unroll 16
F_(SZ/2,UNR_CRZ(crz2,9,16))if(SZ&1){UNR_CRZ(crz,3,4)}return r;}
INLINE W mrot(W x)_(W t=END/3,b=x%t,m=b%3,d=b/3;d+m*(t/3)+(x-b))
P u64 pgsiz;
P W*mem,pat[6];
P void mpstb(void*b,u64 l){mmap(b,l,PROT_READ|PROT_WRITE,MAP_POPULATE|MAP_PRIVATE|MAP_ANON|MAP_FIXED,-1,0);}
P void sigsegvh(int n,siginfo_t*si,void*_) {
void*a=si->si_addr,*ab=(void*)((u64)a&~(pgsiz-1));mpstb(ab, pgsiz);
W* curr=ab;i64 off=(curr-mem)%(END/3);F1(pgsiz,sizeof(W),*curr++=pat[off++%6]);}
P u64 rup(u64 v)_(((v-1)&~(pgsiz-1))+pgsiz)
#define RDS 65536
__attribute__((hot,flatten))int main(int argc, char* argv[]){
pgsiz=sysconf(_SC_PAGESIZE);mem=mmap(NULL,END*sizeof(W),PROT_NONE,MAP_NORESERVE|MAP_PRIVATE|MAP_ANON,-1,0);
struct sigaction act;memset(&act,0,sizeof(struct sigaction));act.sa_flags=SA_SIGINFO;act.sa_sigaction=sigsegvh;sigaction(SIGSEGV,&act,NULL);
FILE*f=fopen(argv[1],"rb");fseek(f,0,SEEK_END);u64 S=ftell(f);rewind(f);u64 szR=rup(S),off=0;mpstb(mem, szR*sizeof(W));char data[RDS];
C W a1_off=94-((END-1)/6-29524)%94,a2_off=94-((END-1)/3-59048)%94;while(S){int am=LKL(S>RDS)?RDS:S;fread(&data,1,am,f);
#pragma GCC unroll 32
F_(am,W w=data[i];mem[off++]=w)S-=am;}for(;off<szR;off++)mem[off]=mcrz(mem[off-1],mem[off-2]);
W n2=mem[off-2],n1=mem[off-1];u64 off2=off;F_(6,W n0=mcrz(n1,n2);pat[off2%6]=n0;n2=n1;n1=n0;off2++)W c=0,a=0,*d=mem;
P C int offs[]={0,((i64)a1_off-(i64)(END/3))%94+94,((i64)a2_off-(i64)(2*(END/3))%94+94)};P C void*j[94];F_(94,j[i]=&&INS_DEF)
#define M(n) j[n]=&&INS_##n;
M(4)M(5)M(23)M(39)M(40)M(62)M(68)M(81)
#define BRA {goto*j[(c+mem[c]+offs[c/(END/3)])%94];}
BRA;
#define NXT mem[c] = \
"SOMEBODY MAKE ME FEEL ALIVE" \
"[hj9>,5z]&gqtyfr$(we4{WP)H-Zn,[%\\3dL+Q;>U!pJS72FhOA1CB6v^=I_0/8|jsb9m<.TVac`uY*MK'X~xDl}REokN:#?G\"i@" \
"AND SHATTER ME"[mem[c]];c++;d++;BRA
INS_4:c=*d;NXT;INS_5:putchar(a);fflush(stdout);NXT;
INS_23:;int CR=getchar();a=CR==EOF?END-1:CR;NXT;INS_39:a=*d=mrot(*d);NXT;INS_40:d=mem+*d;NXT;
INS_62:a=*d=mcrz(a, *d);INS_68:NXT;INS_81:return 0;INS_DEF:NXT;
}
It's an interpreter for rotwidth=19 variant of Malbolge Unshackled (compiled with clang fast20.c -w -O3 -march=native -mtune=native -o fast20 -flto -mllvm -polly -fvisibility=hidden, clang -v yields Debian clang version 11.0.1-2). We feed it with the source code of my project (passed as an argument to the program), temporarily available here (provided hoping that it's possible to reproduce our issue; use 7za to unpack).
Each time I want to run the unit tests, i execute the following shell script:
#!/bin/bash
# XXX: `rsync` is slower
echo "[+] sending test data."
cd kiera-tests && \
tar -czf - * | \
ssh kamila@remote \
"cd ~/malbolgelisp && rm -rf tests && mkdir tests && cd tests && tar -xzf -" && \
cd ..
echo "[+] building essential tools."
ssh kamila@remote "cd ~/malbolgelisp/tests && chmod a+x setup.sh && ./setup.sh"
echo "[+] sending malbolgelisp source code..."
tool/mb_nlib d < lisp.mb | \
pv | gzip -6 | \
ssh kamila@remote \
"gunzip | ~/malbolgelisp/tests/mb_nlib e > ~/malbolgelisp/lisp.mb && vmtouch -vt ~/malbolgelisp/lisp.mb"
echo "[+] running the tests..."
ssh kamila@remote "cd ~/malbolgelisp/tests/ && ./test.sh"
I vmtouch the ~300MB file, so it must have stayed in cache across the runs.
/home/kamila/malbolgelisp/lisp.mb
[OOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOOO] 76711/76711
Files: 1
Directories: 0
Touched Pages: 76711 (299M)
Elapsed: 0.071541 seconds
As we've observed, the cached memory shown by bpytop grows up to 500MiB, which means that the file must have been cached. We also reupload the file each time and it changes significantly.
We tried using Valgrind on the interpreter, but it seems to misbehave under this condition for a yet unknown reason. It's easy to deduce what is happening in the code, though:
pgsiz=sysconf(_SC_PAGESIZE);
mem=mmap(NULL,END*sizeof(W),PROT_NONE,MAP_NORESERVE|MAP_PRIVATE|MAP_ANON,-1,0);
first, entire memory area is mapped.
FILE*f=fopen(argv[1],"rb");fseek(f,0,SEEK_END);u64 S=ftell(f);rewind(f);
u64 szR=rup(S),off=0;mpstb(mem, szR*sizeof(W));
then I query the file size (~300MiB, times sizeof(W) = ~1.2GiB), and map eagerly this amount of memory using mpstb:
P void mpstb(void*b,u64 l){
mmap(b,l,PROT_READ|PROT_WRITE,MAP_POPULATE|MAP_PRIVATE|MAP_ANON|MAP_FIXED,-1,0);}
I considered using mprotect, but in the following parts of the code we execute mpstb fairly often, causing IPIs for TLB shootdowns.
the following bit of code can't be a bottleneck, since aside from the I/O it performs (which is happening on a cached file with a relatively big buffer - RDS = 65536 => should be fast) a bunch of mathematical operations which can't take 7 minutes on one run with the same data and a few seconds on the other run with the same data:
C W a1_off=94-((END-1)/6-29524)%94,a2_off=94-((END-1)/3-59048)%94;
while(S){int am=LKL(S>RDS)?RDS:S;fread(&data,1,am,f);
#pragma GCC unroll 32
F_(am,W w=data[i];mem[off++]=w)S-=am;}
for(;off<szR;off++)mem[off]=mcrz(mem[off-1],mem[off-2]);
We've also noticed that in the following test runner which is executed on the server:
#!/bin/bash
for d in b*; do
for f in $d/*.in; do
echo "[+] $f"
(./fast20 ../lisp.mb $f < $f > $f.aout; diff ${f%%.*}.out $f.aout) &
# sleep 3s
done
for job in `jobs -p`; do
wait $job
done
done
uncommenting the # sleep 3s line makes the allocations much faster, meaning that the Linux kernel simply can't handle a dozen of processes mapping a single gigabyte of memory concurrently. we've also seen these messages pop up during our testing: watchdog: BUG: soft lockup - CPU#34 stuck for 24s! that messed up our bpytop view. Some googling reveals that it's printed when the CPU is stuck for too long in the kernel, which would be yet another argument proving that mmap in this example is ridicously slow.
we've also suspected that it might be caused by memory ballooning on qemu, but disabling it made very little difference.
interestingly enough, all the processes seem to slowly and concurrently allocate memory.
the documentation for the lisp interpreter is available here and it can be used to construct test cases - the simplest one being (+ 2 2).
my question follows - can we do something about this bug? are we missing something? i know that running less processes at a time makes it actually bearable (the runtime drops from 30m to 5m), but if not the allocation performance, the tests could easily finish within 40 seconds, which would be a huge improvement. Is it mmap being inherently slow on Linux when called by multiple processes concurrently?
finally, please let me know if we should provide any further details.
