How to efficiently send large files across a single network connection?

Viewed 305

I'm sending large objects over a network and noticed that using a single network connection is significantly slower than using multiple.

Server code:

use async_std::{
    io::{BufWriter, Write},
    net::TcpListener,
    prelude::*,
    task,
};
use bench_utils::{end_timer, start_timer};
use futures::stream::{FuturesOrdered, StreamExt};

async fn send(buf: &[u8], writer: &mut (impl Write + Unpin)) {
    // Send the message length
    writer.write_all(&(buf.len() as u64).to_le_bytes()).await.unwrap();
    // Send the rest of the message
    writer.write_all(&buf).await.unwrap();
    writer.flush().await.unwrap();
}

fn main() {
    task::block_on(async move {
        let listener = TcpListener::bind("0.0.0.0:8000").await.unwrap();
        let mut incoming = listener.incoming();
        let mut writers = Vec::with_capacity(16);
        for _ in 0..16 {
            let stream = incoming.next().await.unwrap().unwrap();
            writers.push(BufWriter::new(stream))
        };

        let buf = vec![0u8; 1 << 30];
        
        let send_time = start_timer!(|| "Sending buffer across 1 connection");
        send(&buf, &mut writers[0]).await;
        end_timer!(send_time);

        let send_time = start_timer!(|| "Sending buffer across 16 connections");
        writers
            .iter_mut()
            .zip(buf.chunks(buf.len() / 16))
            .map(|(w, chunk)| {
                send(chunk, w)
            })
            .collect::<FuturesOrdered<_>>()
            .collect::<Vec<_>>()
            .await;
        end_timer!(send_time);
    });
}

Client code:

use async_std::{
    io::{BufReader, Read},
    net::TcpStream,
    prelude::*,
    task,
};
use bench_utils::{end_timer, start_timer};
use futures::stream::{FuturesOrdered, StreamExt};

async fn recv(reader: &mut (impl Read + Unpin)) {
    // Read the message length
    let mut len_buf = [0u8; 8];
    reader.read_exact(&mut len_buf).await.unwrap();
    let len: u64 = u64::from_le_bytes(len_buf);
    // Read the rest of the message
    let mut buf = vec![0u8; usize::try_from(len).unwrap()];
    reader.read_exact(&mut buf[..]).await.unwrap();
}

fn main() {
    let host = &std::env::args().collect::<Vec<_>>()[1];
    task::block_on(async move {
        let mut readers = Vec::with_capacity(16);
        for _ in 0..16 {
            let stream = TcpStream::connect(host).await.unwrap();
            readers.push(BufReader::new(stream));
        }

        let read_time = start_timer!(|| "Reading buffer from 1 connection");
        recv(&mut readers[0]).await;
        end_timer!(read_time);

        let read_time = start_timer!(|| "Reading buffer from 16 connections");
        readers 
        .iter_mut()
        .map(|r| recv(r))
        .collect::<FuturesOrdered<_>>()
        .collect::<Vec<_>>()
        .await;
        end_timer!(read_time);
    });
}

Server result:

Start:   Sending buffer across 1 connection
End:     Sending buffer across 1 connection....................................55.134s
Start:   Sending buffer across 16 connections
End:     Sending buffer across 16 connections..................................4.19s

Client result:

Start:   Reading buffer from 1 connection
End:     Reading buffer from 1 connection......................................55.396s
Start:   Reading buffer from 16 connections
End:     Reading buffer from 16 connections....................................3.914s

I am assuming that this difference is due to the sending connection having to wait for an ACK when the TCP buffer is filled (both machines have TCP window scaling enabled)? It doesn't appear that Rust provides an API to modify the size of these things.

Is there anyway to achieve similar throughput on a single connection? It seems annoying to have to pass around multiple since all of this is going through a single network interface anyways.

0 Answers
Related