#! /usr/bin/bash

# Parent cgroup of our containers' cgroups
PARENT_CGROUP=containers

hostname="$1"
image="$2"
shift 2

memory_MB="$1"
cpu_perc="$2"
shift 2

ipaddr=$1
shift 1

# Convert memory limit from MB to B
memory=$(expr "$memory_MB" '*' 1024 '*' 1024)

# Flags for the unshare command, completed one namespace at a time
UNSHARE_FLAGS="--user --map-root-user\
    --uts\
    --pid --fork\
    --net\
    --mount"

# Create a named pipe for cross-namespace communication
mkfifo "$hostname.pipe"
# Redirect file descriptor 3 to the pipe
exec 3<>"$hostname.pipe"

# The name of the container's FS directory is built from the given hostname and
# image name
contfs="$image-$hostname"
# Use --parents to ignore error when directories exist
mkdir --parents "$contfs"/{.diff,.workdir,run}
# Mount the overlay FS under the container's directory, in "run"
sudo mount -t overlay overlay \
    -o lowerdir="$image",upperdir="$contfs/.diff",workdir="$contfs/.workdir" \
    "$contfs/run"

memory_cgroup="/sys/fs/cgroup/memory/$PARENT_CGROUP/$hostname"
mkdir "$memory_cgroup"
echo $memory > "$memory_cgroup/memory.limit_in_bytes"

cpu_cgroup="/sys/fs/cgroup/cpu/$PARENT_CGROUP/$hostname"
mkdir "$cpu_cgroup"
echo $(expr "$cpu_perc" '*' "$(cat "$cpu_cgroup/cpu.cfs_period_us")" / 100) > \
    "$cpu_cgroup/cpu.cfs_quota_us"

echo $$ > "$memory_cgroup/cgroup.procs"
echo $$ > "$cpu_cgroup/cgroup.procs"

# Note that we fork to background
unshare $UNSHARE_FLAGS ./continit.sh "$hostname" "$contfs/run" "$@" &
continit_pid=$!

# vethOutXXX is the end on the host, vethInXXX is the end in the container
sudo ip link add "vethOut$hostname" type veth peer name "vethIn$hostname"
sudo ip link set "vethOut$hostname" up
# Insert the container's end inside the net namespace
sudo ip link set "vethIn$hostname" netns $continit_pid
sudo ip link set "vethOut$hostname" master contbr0

echo "vethIn$hostname" "$ipaddr" >&3
rm "$hostname.pipe"

