2014-02-20 00:50:10 +00:00
|
|
|
// +build linux
|
|
|
|
|
2014-02-21 02:27:42 +00:00
|
|
|
package nsinit
|
2014-02-19 00:56:11 +00:00
|
|
|
|
|
|
|
import (
|
|
|
|
"fmt"
|
2014-02-19 07:13:36 +00:00
|
|
|
"github.com/dotcloud/docker/pkg/system"
|
2014-02-19 00:56:11 +00:00
|
|
|
"os"
|
|
|
|
"path/filepath"
|
|
|
|
"syscall"
|
|
|
|
)
|
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// default mount point flags
|
2014-02-20 00:40:36 +00:00
|
|
|
const defaultMountFlags = syscall.MS_NOEXEC | syscall.MS_NOSUID | syscall.MS_NODEV
|
2014-02-19 00:56:11 +00:00
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// setupNewMountNamespace is used to initialize a new mount namespace for an new
|
|
|
|
// container in the rootfs that is specified.
|
|
|
|
//
|
|
|
|
// There is no need to unmount the new mounts because as soon as the mount namespace
|
|
|
|
// is no longer in use, the mounts will be removed automatically
|
2014-02-19 22:33:25 +00:00
|
|
|
func setupNewMountNamespace(rootfs, console string, readonly bool) error {
|
2014-02-20 06:43:40 +00:00
|
|
|
// mount as slave so that the new mounts do not propagate to the host
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount("", "/", "", syscall.MS_SLAVE|syscall.MS_REC, ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mounting / as slave %s", err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount(rootfs, rootfs, "bind", syscall.MS_BIND|syscall.MS_REC, ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mouting %s as bind %s", rootfs, err)
|
|
|
|
}
|
|
|
|
if readonly {
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount(rootfs, rootfs, "bind", syscall.MS_BIND|syscall.MS_REMOUNT|syscall.MS_RDONLY|syscall.MS_REC, ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mounting %s as readonly %s", rootfs, err)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
if err := mountSystem(rootfs); err != nil {
|
|
|
|
return fmt.Errorf("mount system %s", err)
|
|
|
|
}
|
|
|
|
if err := copyDevNodes(rootfs); err != nil {
|
|
|
|
return fmt.Errorf("copy dev nodes %s", err)
|
|
|
|
}
|
2014-02-27 01:21:09 +00:00
|
|
|
if err := setupLoopbackDevices(rootfs); err != nil {
|
|
|
|
return fmt.Errorf("setup loopback devices %s", err)
|
|
|
|
}
|
2014-02-19 00:56:11 +00:00
|
|
|
if err := setupDev(rootfs); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
2014-02-21 02:05:40 +00:00
|
|
|
if console != "" {
|
|
|
|
if err := setupPtmx(rootfs, console); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
2014-02-19 00:56:11 +00:00
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Chdir(rootfs); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("chdir into %s %s", rootfs, err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount(rootfs, "/", "", syscall.MS_MOVE, ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mount move %s into / %s", rootfs, err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Chroot("."); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("chroot . %s", err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Chdir("/"); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("chdir / %s", err)
|
|
|
|
}
|
|
|
|
|
2014-02-19 07:13:36 +00:00
|
|
|
system.Umask(0022)
|
2014-02-19 00:56:11 +00:00
|
|
|
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// copyDevNodes mknods the hosts devices so the new container has access to them
|
2014-02-19 00:56:11 +00:00
|
|
|
func copyDevNodes(rootfs string) error {
|
2014-02-19 07:13:36 +00:00
|
|
|
oldMask := system.Umask(0000)
|
|
|
|
defer system.Umask(oldMask)
|
2014-02-19 00:56:11 +00:00
|
|
|
|
|
|
|
for _, node := range []string{
|
|
|
|
"null",
|
|
|
|
"zero",
|
|
|
|
"full",
|
|
|
|
"random",
|
|
|
|
"urandom",
|
|
|
|
"tty",
|
|
|
|
} {
|
2014-02-27 01:21:09 +00:00
|
|
|
if err := copyDevNode(rootfs, node); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return err
|
|
|
|
}
|
2014-02-27 01:21:09 +00:00
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
|
|
|
func setupLoopbackDevices(rootfs string) error {
|
|
|
|
for i := 0; ; i++ {
|
2014-02-19 00:56:11 +00:00
|
|
|
var (
|
2014-02-27 01:21:09 +00:00
|
|
|
device = fmt.Sprintf("loop%d", i)
|
|
|
|
source = filepath.Join("/dev", device)
|
|
|
|
dest = filepath.Join(rootfs, "dev", device)
|
2014-02-19 00:56:11 +00:00
|
|
|
)
|
2014-02-27 01:21:09 +00:00
|
|
|
|
|
|
|
if _, err := os.Stat(source); err != nil {
|
|
|
|
if !os.IsNotExist(err) {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
if _, err := os.Stat(dest); err == nil {
|
|
|
|
os.Remove(dest)
|
2014-02-19 00:56:11 +00:00
|
|
|
}
|
2014-02-27 01:21:09 +00:00
|
|
|
f, err := os.Create(dest)
|
|
|
|
if err != nil {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
f.Close()
|
|
|
|
if err := system.Mount(source, dest, "none", syscall.MS_BIND, ""); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
|
|
|
func copyDevNode(rootfs, node string) error {
|
|
|
|
stat, err := os.Stat(filepath.Join("/dev", node))
|
|
|
|
if err != nil {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
var (
|
|
|
|
dest = filepath.Join(rootfs, "dev", node)
|
|
|
|
st = stat.Sys().(*syscall.Stat_t)
|
|
|
|
)
|
|
|
|
if err := system.Mknod(dest, st.Mode, int(st.Rdev)); err != nil && !os.IsExist(err) {
|
|
|
|
return fmt.Errorf("copy %s %s", node, err)
|
2014-02-19 00:56:11 +00:00
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// setupDev symlinks the current processes pipes into the
|
|
|
|
// appropriate destination on the containers rootfs
|
2014-02-19 00:56:11 +00:00
|
|
|
func setupDev(rootfs string) error {
|
|
|
|
for _, link := range []struct {
|
|
|
|
from string
|
|
|
|
to string
|
|
|
|
}{
|
|
|
|
{"/proc/kcore", "/dev/core"},
|
|
|
|
{"/proc/self/fd", "/dev/fd"},
|
|
|
|
{"/proc/self/fd/0", "/dev/stdin"},
|
|
|
|
{"/proc/self/fd/1", "/dev/stdout"},
|
|
|
|
{"/proc/self/fd/2", "/dev/stderr"},
|
|
|
|
} {
|
|
|
|
dest := filepath.Join(rootfs, link.to)
|
|
|
|
if err := os.Remove(dest); err != nil && !os.IsNotExist(err) {
|
|
|
|
return fmt.Errorf("remove %s %s", dest, err)
|
|
|
|
}
|
|
|
|
if err := os.Symlink(link.from, dest); err != nil {
|
|
|
|
return fmt.Errorf("symlink %s %s", dest, err)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// setupConsole ensures that the container has a proper /dev/console setup
|
2014-02-19 00:56:11 +00:00
|
|
|
func setupConsole(rootfs, console string) error {
|
2014-02-19 07:13:36 +00:00
|
|
|
oldMask := system.Umask(0000)
|
|
|
|
defer system.Umask(oldMask)
|
2014-02-19 00:56:11 +00:00
|
|
|
|
|
|
|
stat, err := os.Stat(console)
|
|
|
|
if err != nil {
|
|
|
|
return fmt.Errorf("stat console %s %s", console, err)
|
|
|
|
}
|
2014-02-20 00:40:36 +00:00
|
|
|
var (
|
|
|
|
st = stat.Sys().(*syscall.Stat_t)
|
|
|
|
dest = filepath.Join(rootfs, "dev/console")
|
|
|
|
)
|
2014-02-19 00:56:11 +00:00
|
|
|
if err := os.Remove(dest); err != nil && !os.IsNotExist(err) {
|
|
|
|
return fmt.Errorf("remove %s %s", dest, err)
|
|
|
|
}
|
|
|
|
if err := os.Chmod(console, 0600); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
if err := os.Chown(console, 0, 0); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mknod(dest, (st.Mode&^07777)|0600, int(st.Rdev)); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mknod %s %s", dest, err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount(console, dest, "bind", syscall.MS_BIND, ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("bind %s to %s %s", console, dest, err)
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
|
|
|
// mountSystem sets up linux specific system mounts like sys, proc, shm, and devpts
|
|
|
|
// inside the mount namespace
|
|
|
|
func mountSystem(rootfs string) error {
|
2014-02-19 07:13:36 +00:00
|
|
|
for _, m := range []struct {
|
2014-02-19 00:56:11 +00:00
|
|
|
source string
|
|
|
|
path string
|
|
|
|
device string
|
|
|
|
flags int
|
|
|
|
data string
|
|
|
|
}{
|
2014-02-20 00:40:36 +00:00
|
|
|
{source: "proc", path: filepath.Join(rootfs, "proc"), device: "proc", flags: defaultMountFlags},
|
|
|
|
{source: "sysfs", path: filepath.Join(rootfs, "sys"), device: "sysfs", flags: defaultMountFlags},
|
2014-02-19 00:56:11 +00:00
|
|
|
{source: "tmpfs", path: filepath.Join(rootfs, "dev"), device: "tmpfs", flags: syscall.MS_NOSUID | syscall.MS_STRICTATIME, data: "mode=755"},
|
2014-02-20 00:40:36 +00:00
|
|
|
{source: "shm", path: filepath.Join(rootfs, "dev", "shm"), device: "tmpfs", flags: defaultMountFlags, data: "mode=1777"},
|
2014-02-19 00:56:11 +00:00
|
|
|
{source: "devpts", path: filepath.Join(rootfs, "dev", "pts"), device: "devpts", flags: syscall.MS_NOSUID | syscall.MS_NOEXEC, data: "newinstance,ptmxmode=0666,mode=620,gid=5"},
|
|
|
|
{source: "tmpfs", path: filepath.Join(rootfs, "run"), device: "tmpfs", flags: syscall.MS_NOSUID | syscall.MS_NODEV | syscall.MS_STRICTATIME, data: "mode=755"},
|
2014-02-19 07:13:36 +00:00
|
|
|
} {
|
2014-02-19 00:56:11 +00:00
|
|
|
if err := os.MkdirAll(m.path, 0755); err != nil && !os.IsExist(err) {
|
|
|
|
return fmt.Errorf("mkdirall %s %s", m.path, err)
|
|
|
|
}
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Mount(m.source, m.path, m.device, uintptr(m.flags), m.data); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return fmt.Errorf("mounting %s into %s %s", m.source, m.path, err)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
2014-02-20 06:43:40 +00:00
|
|
|
// setupPtmx adds a symlink to pts/ptmx for /dev/ptmx and
|
|
|
|
// finishes setting up /dev/console
|
|
|
|
func setupPtmx(rootfs, console string) error {
|
|
|
|
ptmx := filepath.Join(rootfs, "dev/ptmx")
|
|
|
|
if err := os.Remove(ptmx); err != nil && !os.IsNotExist(err) {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
if err := os.Symlink("pts/ptmx", ptmx); err != nil {
|
|
|
|
return fmt.Errorf("symlink dev ptmx %s", err)
|
|
|
|
}
|
|
|
|
if err := setupConsole(rootfs, console); err != nil {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
|
|
|
// remountProc is used to detach and remount the proc filesystem
|
|
|
|
// commonly needed with running a new process inside an existing container
|
2014-02-19 00:56:11 +00:00
|
|
|
func remountProc() error {
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Unmount("/proc", syscall.MNT_DETACH); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return err
|
|
|
|
}
|
2014-02-20 00:40:36 +00:00
|
|
|
if err := system.Mount("proc", "/proc", "proc", uintptr(defaultMountFlags), ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return err
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|
|
|
|
|
|
|
|
func remountSys() error {
|
2014-02-19 07:13:36 +00:00
|
|
|
if err := system.Unmount("/sys", syscall.MNT_DETACH); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
if err != syscall.EINVAL {
|
|
|
|
return err
|
|
|
|
}
|
|
|
|
} else {
|
2014-02-20 00:40:36 +00:00
|
|
|
if err := system.Mount("sysfs", "/sys", "sysfs", uintptr(defaultMountFlags), ""); err != nil {
|
2014-02-19 00:56:11 +00:00
|
|
|
return err
|
|
|
|
}
|
|
|
|
}
|
|
|
|
return nil
|
|
|
|
}
|