// Copyright (c) Contributors to the Apptainer project, established as
//   Apptainer a Series of LF Projects LLC.
//   For website terms of use, trademark policy, privacy policy and other
//   project policies see https://lfprojects.org/policies
// Copyright (c) 2018-2022, Sylabs Inc. All rights reserved.
// This software is licensed under a 3-clause BSD license. Please consult the
// LICENSE.md file distributed with the sources of this project regarding your
// rights to use or distribute this software.

package oci

import (
	"bufio"
	"context"
	"encoding/json"
	"fmt"
	"net"
	"net/rpc"
	"os"
	"path/filepath"
	"strings"
	"syscall"
	"time"

	"github.com/apptainer/apptainer/internal/pkg/cgroups"
	"github.com/apptainer/apptainer/internal/pkg/instance"
	"github.com/apptainer/apptainer/internal/pkg/runtime/engine/oci/rpc/client"
	"github.com/apptainer/apptainer/internal/pkg/util/fs"
	"github.com/apptainer/apptainer/internal/pkg/util/fs/mount"
	"github.com/apptainer/apptainer/pkg/ociruntime"
	"github.com/apptainer/apptainer/pkg/sylog"
	"github.com/apptainer/apptainer/pkg/util/fs/proc"
	"github.com/apptainer/apptainer/pkg/util/namespaces"
	"github.com/apptainer/apptainer/pkg/util/sysctl"
	"github.com/apptainer/apptainer/pkg/util/unix"
	specs "github.com/opencontainers/runtime-spec/specs-go"
)

var symlinkDevices = []struct {
	old string
	new string
}{
	{"/proc/self/fd", "/dev/fd"},
	{"/proc/kcore", "/dev/core"},
	{"pts/ptmx", "/dev/ptmx"},
	{"/proc/self/fd/0", "/dev/stdin"},
	{"/proc/self/fd/1", "/dev/stdout"},
	{"/proc/self/fd/2", "/dev/stderr"},
}

type device struct {
	major int64
	minor int64
	path  string
	mode  os.FileMode
	uid   int
	gid   int
}

var devices = []device{
	{1, 7, "/dev/full", syscall.S_IFCHR | 0o666, 0, 0},
	{1, 3, "/dev/null", syscall.S_IFCHR | 0o666, 0, 0},
	{1, 8, "/dev/random", syscall.S_IFCHR | 0o666, 0, 0},
	{5, 0, "/dev/tty", syscall.S_IFCHR | 0o666, 0, 0},
	{1, 9, "/dev/urandom", syscall.S_IFCHR | 0o666, 0, 0},
	{1, 5, "/dev/zero", syscall.S_IFCHR | 0o666, 0, 0},
}

var cgroupDevices = []specs.LinuxDeviceCgroup{
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(1),
		Minor:  cgroups.Int64ptr(7),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(1),
		Minor:  cgroups.Int64ptr(3),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(1),
		Minor:  cgroups.Int64ptr(8),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(5),
		Minor:  cgroups.Int64ptr(0),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(1),
		Minor:  cgroups.Int64ptr(9),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(1),
		Minor:  cgroups.Int64ptr(5),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(136),
		Minor:  cgroups.Int64ptr(-1),
		Access: "rwm",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(5),
		Minor:  cgroups.Int64ptr(1),
		Access: "rw",
	},
	{
		Allow:  true,
		Type:   "c",
		Major:  cgroups.Int64ptr(5),
		Minor:  cgroups.Int64ptr(2),
		Access: "rw",
	},
}

type container struct {
	engine             *EngineOperations
	rpcOps             *client.RPC
	rootfs             string
	rpcRoot            string
	userNS             bool
	utsNS              bool
	mntNS              bool
	devIndex           int
	cgroupV1MountIndex int
}

var statusChan = make(chan string, 1)

// CreateContainer is called from master process to prepare container
// environment, e.g. perform mount operations, etc.
//
// Additional privileges required for setup may be gained when running
// in suid flow. However, when a user namespace is requested and it is not
// a hybrid workflow (e.g. fakeroot), then there is no privileged saved uid
// and thus no additional privileges can be gained.
//
// Specifically in oci engine, no additional privileges are gained. Container
// setup (e.g. mount operations) where privileges may be required is performed
// by calling RPC server methods (see internal/app/starter/rpc_linux.go for details).
//
// However, most likely this still will be executed as root since `apptainer oci`
// command set requires privileged execution.
//
//nolint:maintidx
func (e *EngineOperations) CreateContainer(_ context.Context, pid int, rpcConn net.Conn) error {
	var err error

	if e.CommonConfig.EngineName != Name {
		return fmt.Errorf("engineName configuration doesn't match runtime name")
	}

	rpcOps := &client.RPC{}
	rpcOps.Client = rpc.NewClient(rpcConn)
	rpcOps.Name = e.CommonConfig.EngineName

	if rpcOps.Client == nil {
		return fmt.Errorf("failed to initialize RPC client")
	}

	if err := e.createState(pid); err != nil {
		return err
	}

	rootfs := e.EngineConfig.OciConfig.Root.Path

	if !filepath.IsAbs(rootfs) {
		rootfs = filepath.Join(e.EngineConfig.GetBundlePath(), rootfs)
	}

	resolvedRootfs, err := filepath.EvalSymlinks(rootfs)
	if err != nil {
		return fmt.Errorf("failed to resolve %s path: %s", rootfs, err)
	}

	c := &container{
		engine:             e,
		rpcOps:             rpcOps,
		rootfs:             resolvedRootfs,
		rpcRoot:            fmt.Sprintf("/proc/%d/root", pid),
		cgroupV1MountIndex: -1,
		devIndex:           -1,
	}

	for _, ns := range e.EngineConfig.OciConfig.Linux.Namespaces {
		switch ns.Type {
		case specs.UserNamespace:
			c.userNS = true
		case specs.UTSNamespace:
			c.utsNS = true
		case specs.MountNamespace:
			c.mntNS = true
		}
	}

	p := &mount.Points{}
	if e.EngineConfig.OciConfig.Linux.MountLabel != "" {
		if err := p.SetContext(e.EngineConfig.OciConfig.Linux.MountLabel); err != nil {
			return err
		}
	}

	system := &mount.System{Points: p, Mount: c.mount}

	for i, point := range e.EngineConfig.OciConfig.Config.Mounts {
		// A cgroup v1 mount point will be intercepted and handled separately in c.addCgroups(...)
		if point.Type == "cgroup" {
			c.cgroupV1MountIndex = i
			continue
		}
		// dev creation
		if point.Destination == "/dev" && point.Type == "tmpfs" {
			c.devIndex = i
		}
	}

	if err := c.addDevices(system); err != nil {
		return err
	}

	if err := c.addCgroups(pid, system); err != nil {
		return err
	}

	// import OCI mount spec
	if err := system.Points.ImportFromSpec(e.EngineConfig.OciConfig.Config.Mounts); err != nil {
		return err
	}

	if err := c.addRootfsMount(system); err != nil {
		return err
	}

	if err := system.RunAfterTag(mount.KernelTag, c.addDefaultDevices); err != nil {
		return err
	}

	if err := system.RunAfterTag(mount.KernelTag, c.addAllPaths); err != nil {
		return err
	}

	if err := proc.SetOOMScoreAdj(pid, e.EngineConfig.OciConfig.Process.OOMScoreAdj); err != nil {
		return err
	}

	if err := namespaces.Enter(pid, "ipc"); err != nil {
		return err
	}
	if err := namespaces.Enter(pid, "net"); err != nil {
		return err
	}

	for key, value := range e.EngineConfig.OciConfig.Linux.Sysctl {
		if err := sysctl.Set(key, value); err != nil {
			return err
		}
	}

	if err := namespaces.Enter(os.Getpid(), "ipc"); err != nil {
		return err
	}
	if err := namespaces.Enter(os.Getpid(), "net"); err != nil {
		return err
	}

	sylog.Debugf("Mount all")
	if err := system.MountAll(); err != nil {
		return err
	}

	if c.utsNS && e.EngineConfig.OciConfig.Hostname != "" {
		if _, err := rpcOps.SetHostname(e.EngineConfig.OciConfig.Hostname); err != nil {
			return err
		}
	}

	// update namespaces configuration path
	namespaces := []struct {
		nstype       string
		ns           specs.LinuxNamespaceType
		checkEnabled bool
	}{
		{"pid", specs.PIDNamespace, false},
		{"uts", specs.UTSNamespace, false},
		{"ipc", specs.IPCNamespace, false},
		{"mnt", specs.MountNamespace, false},
		{"cgroup", specs.CgroupNamespace, false},
		{"net", specs.NetworkNamespace, false},
		{"user", specs.UserNamespace, true},
	}

	path := fmt.Sprintf("/proc/%d/ns", pid)

	for _, n := range namespaces {
		has, err := proc.HasNamespace(pid, n.nstype)
		if err == nil && (has || n.checkEnabled) {
			enabled := false
			if n.checkEnabled {
				if e.EngineConfig.OciConfig.Linux != nil {
					for _, namespace := range e.EngineConfig.OciConfig.Linux.Namespaces {
						if n.ns == namespace.Type {
							enabled = true
							break
						}
					}
				}
			}
			if has || enabled {
				nspath := filepath.Join(path, n.nstype)
				e.EngineConfig.OciConfig.AddOrReplaceLinuxNamespace(n.ns, nspath)
			}
		} else if err != nil {
			return fmt.Errorf("failed to check %s root and container namespace: %s", n.ns, err)
		}
	}

	method := "pivot"
	if !c.mntNS {
		method = "chroot"
	}

	_, err = rpcOps.Chroot(c.rootfs, method)
	if err != nil {
		return fmt.Errorf("chroot failed: %s", err)
	}

	if e.EngineConfig.SlavePts != -1 {
		if err := syscall.Close(e.EngineConfig.SlavePts); err != nil {
			return fmt.Errorf("failed to close slave part: %s", err)
		}
	}
	if e.EngineConfig.OutputStreams[0] != -1 {
		if err := syscall.Close(e.EngineConfig.OutputStreams[1]); err != nil {
			return fmt.Errorf("failed to close write output stream: %s", err)
		}
	}
	if e.EngineConfig.ErrorStreams[0] != -1 {
		if err := syscall.Close(e.EngineConfig.ErrorStreams[1]); err != nil {
			return fmt.Errorf("failed to close write error stream: %s", err)
		}
	}
	if e.EngineConfig.InputStreams[0] != -1 {
		if err := syscall.Close(e.EngineConfig.InputStreams[1]); err != nil {
			return fmt.Errorf("failed to close write input stream: %s", err)
		}
	}

	return nil
}

func (e *EngineOperations) createState(pid int) error {
	e.EngineConfig.Lock()
	defer e.EngineConfig.Unlock()

	name := e.CommonConfig.ContainerID

	file, err := instance.Add(name, instance.OciSubDir)
	if err != nil {
		return err
	}

	e.EngineConfig.State.Version = specs.Version
	e.EngineConfig.State.Bundle = e.EngineConfig.GetBundlePath()
	e.EngineConfig.State.ID = e.CommonConfig.ContainerID
	e.EngineConfig.State.Pid = pid
	e.EngineConfig.State.Status = ociruntime.Creating
	e.EngineConfig.State.Annotations = e.EngineConfig.OciConfig.Annotations

	file.Config, err = json.Marshal(e.CommonConfig)
	if err != nil {
		return err
	}

	file.User = "root"
	file.Pid = pid
	file.PPid = os.Getpid()
	file.Image = filepath.Join(e.EngineConfig.GetBundlePath(), e.EngineConfig.OciConfig.Root.Path)

	if err := file.Update(); err != nil {
		return err
	}

	socketPath := e.EngineConfig.SyncSocket

	if socketPath != "" {
		data, err := json.Marshal(e.EngineConfig.State)
		if err != nil {
			sylog.Warningf("failed to marshal state data: %s", err)
		} else if err := unix.WriteSocket(socketPath, data); err != nil {
			sylog.Warningf("%s", err)
		}
	}

	return nil
}

func (e *EngineOperations) updateState(status string) error {
	e.EngineConfig.Lock()
	defer e.EngineConfig.Unlock()

	file, err := instance.Get(e.CommonConfig.ContainerID, instance.OciSubDir)
	if err != nil {
		return err
	}
	// do nothing if already stopped
	if e.EngineConfig.State.Status == ociruntime.Stopped {
		return nil
	}
	oldStatus := e.EngineConfig.State.Status
	e.EngineConfig.State.Status = specs.ContainerState(status)

	t := time.Now().UnixNano()

	switch status {
	case ociruntime.Created:
		if e.EngineConfig.State.CreatedAt == nil {
			e.EngineConfig.State.CreatedAt = &t
		}
	case ociruntime.Running:
		if e.EngineConfig.State.StartedAt == nil {
			e.EngineConfig.State.StartedAt = &t
		}
	case ociruntime.Stopped:
		if e.EngineConfig.State.FinishedAt == nil {
			e.EngineConfig.State.FinishedAt = &t
		}
	}

	file.Config, err = json.Marshal(e.CommonConfig)
	if err != nil {
		return err
	}

	if err := file.Update(); err != nil {
		return err
	}

	socketPath := e.EngineConfig.SyncSocket

	if socketPath != "" {
		data, err := json.Marshal(e.EngineConfig.State)
		if err != nil {
			sylog.Warningf("failed to marshal state data: %s", err)
		} else if err := unix.WriteSocket(socketPath, data); err != nil {
			sylog.Warningf("%s", err)
		}
	}

	// send running or stopped status right after container creation
	// to notify that container process started
	if statusChan != nil && oldStatus == ociruntime.Created &&
		(status == ociruntime.Running || status == ociruntime.Stopped) {
		statusChan <- status
	}
	return nil
}

// one shot function to wait on running or stopped status
func (e *EngineOperations) waitStatusUpdate() {
	if statusChan == nil {
		return
	}
	// block until status update is sent
	<-statusChan
	// close channel and set it to nil
	close(statusChan)
	statusChan = nil
}

func (c *container) addCgroups(pid int, system *mount.System) error {
	name := c.engine.CommonConfig.ContainerID
	resources := c.engine.EngineConfig.OciConfig.Linux.Resources
	systemd := c.engine.EngineConfig.GetSystemdCgroups()
	cgroupsPath := c.engine.EngineConfig.OciConfig.Linux.CgroupsPath

	if !systemd && !filepath.IsAbs(cgroupsPath) {
		if cgroupsPath == "" {
			cgroupsPath = filepath.Join("/apptainer-oci", name)
		} else {
			cgroupsPath = filepath.Join("/", cgroupsPath)
		}
	}

	if systemd && cgroupsPath == "" {
		cgroupsPath = "system.slice:apptainer-oci:" + name
	}

	c.engine.EngineConfig.OciConfig.Linux.CgroupsPath = cgroupsPath

	manager, err := cgroups.NewManagerWithSpec(resources, pid, cgroupsPath, systemd)
	if err != nil {
		return fmt.Errorf("failed to apply cgroups resources restriction: %s", err)
	}

	// If a mount point exists for a cgroup v1 hierarchy we will handle it here.
	// This is not necessary for cgroups v2 - as the unified hierarchy will be handled with a simple bind.
	if c.cgroupV1MountIndex >= 0 {
		m := c.engine.EngineConfig.OciConfig.Config.Mounts[c.cgroupV1MountIndex]
		c.engine.EngineConfig.OciConfig.Config.Mounts = append(
			c.engine.EngineConfig.OciConfig.Config.Mounts[:c.cgroupV1MountIndex],
			c.engine.EngineConfig.OciConfig.Config.Mounts[c.cgroupV1MountIndex+1:]...,
		)

		cgroupRootPath, err := manager.GetCgroupRootPath()
		if err != nil {
			return fmt.Errorf("failed to determine cgroup root path: %w", err)
		}

		flags, opt := mount.ConvertOptions(m.Options)
		options := strings.Join(opt, ",")

		readOnly := false
		if flags&syscall.MS_RDONLY != 0 {
			readOnly = true
			flags &^= uintptr(syscall.MS_RDONLY)
		}

		hasMode := false
		for _, o := range opt {
			if strings.HasPrefix(o, "mode=") {
				hasMode = true
				break
			}
		}
		if !hasMode {
			options += ",mode=755"
		}

		if err := system.Points.AddFS(mount.OtherTag, m.Destination, "tmpfs", flags, options); err != nil {
			return err
		}

		createSymlinks := func(*mount.System) error {
			cgroupPath := filepath.Join(c.rpcRoot, c.rootfs, m.Destination)
			if _, err := os.Stat(filepath.Join(cgroupPath, "cpu")); err != nil && os.IsNotExist(err) {
				if err := c.rpcOps.Symlink("cpu,cpuacct", filepath.Join(c.rootfs, m.Destination, "cpu")); err != nil {
					return err
				}
				if err := c.rpcOps.Symlink("cpu,cpuacct", filepath.Join(c.rootfs, m.Destination, "cpuacct")); err != nil {
					return err
				}
			}

			if _, err := os.Stat(filepath.Join(cgroupPath, "net_cls")); err != nil && os.IsNotExist(err) {
				if err := c.rpcOps.Symlink("net_cls,net_prio", filepath.Join(c.rootfs, m.Destination, "net_cls")); err != nil {
					return err
				}
				if err := c.rpcOps.Symlink("net_cls,net_prio", filepath.Join(c.rootfs, m.Destination, "net_prio")); err != nil {
					return err
				}
			}
			return nil
		}

		if err := system.RunAfterTag(mount.OtherTag, createSymlinks); err != nil {
			return err
		}

		f, err := os.Open(fmt.Sprintf("/proc/%d/cgroup", pid))
		if err != nil {
			return err
		}
		defer f.Close()

		flags |= uintptr(syscall.MS_BIND)
		if readOnly {
			flags |= syscall.MS_RDONLY
		}

		scanner := bufio.NewScanner(f)
		for scanner.Scan() {
			cgroupLine := strings.Split(scanner.Text(), ":")
			if strings.HasPrefix(cgroupLine[1], "name=") {
				cgroupLine[1] = strings.Replace(cgroupLine[1], "name=", "", 1)
			}
			if cgroupLine[1] != "" {
				source := filepath.Join(cgroupRootPath, cgroupLine[1], cgroupLine[2])
				dest := filepath.Join(m.Destination, cgroupLine[1])
				if err := system.Points.AddBind(mount.OtherTag, source, dest, flags); err != nil {
					return err
				}
				if readOnly {
					if err := system.Points.AddRemount(mount.OtherTag, dest, flags); err != nil {
						return err
					}
				}
			}
		}

		if readOnly {
			if err := system.Points.AddRemount(mount.FinalTag, m.Destination, flags); err != nil {
				return err
			}
		}
	}

	c.engine.EngineConfig.Cgroups = manager

	return nil
}

func (c *container) addAllPaths(system *mount.System) error {
	// add masked path
	if err := c.addMaskedPathsMount(system); err != nil {
		return err
	}

	// add read-only path
	if !c.userNS {
		if err := c.addReadonlyPathsMount(system); err != nil {
			return err
		}
	}

	return nil
}

func (c *container) addRootfsMount(system *mount.System) error {
	flags := uintptr(syscall.MS_BIND)

	if c.engine.EngineConfig.OciConfig.Root.Readonly {
		sylog.Debugf("Mounted read-only")
		flags |= syscall.MS_RDONLY
	}

	parentRootfs, err := proc.ParentMount(c.rootfs)
	if err != nil {
		return err
	}

	sylog.Debugf("Parent rootfs: %s", parentRootfs)

	if err := c.rpcOps.Mount("", parentRootfs, "", syscall.MS_PRIVATE, ""); err != nil {
		return err
	}
	if err := system.Points.AddBind(mount.RootfsTag, c.rootfs, c.rootfs, flags); err != nil {
		return err
	}
	if flags&syscall.MS_RDONLY != 0 {
		return system.Points.AddRemount(mount.FinalTag, c.rootfs, flags)
	}

	return nil
}

func (c *container) addDefaultDevices(system *mount.System) error {
	oldmask := syscall.Umask(0)
	defer syscall.Umask(oldmask)

	rootfsPath := filepath.Join(c.rpcRoot, c.rootfs)

	devPath := filepath.Join(rootfsPath, fs.EvalRelative("/dev", rootfsPath))
	if _, err := os.Lstat(devPath); os.IsNotExist(err) {
		if err := os.Mkdir(devPath, 0o755); err != nil {
			return err
		}
	}

	for _, symlink := range symlinkDevices {
		path := filepath.Join(rootfsPath, symlink.new)
		if _, err := os.Lstat(path); os.IsNotExist(err) {
			if c.userNS {
				path = filepath.Join(c.rootfs, symlink.new)
				if err := c.rpcOps.Symlink(symlink.old, path); err != nil {
					return err
				}
			} else {
				if err := os.Symlink(symlink.old, path); err != nil {
					return err
				}
			}
		}
	}

	if c.engine.EngineConfig.OciConfig.Process.Terminal {
		path := filepath.Join(rootfsPath, "dev", "console")
		if _, err := os.Lstat(path); os.IsNotExist(err) {
			if c.userNS {
				if _, err := c.rpcOps.Touch(filepath.Join(c.rootfs, "dev", "console")); err != nil {
					return err
				}
			} else {
				if err := fs.Touch(path); err != nil {
					return err
				}
			}
			path = fmt.Sprintf("/proc/self/fd/%d", c.engine.EngineConfig.SlavePts)
			console, err := os.Readlink(path)
			if err != nil {
				return err
			}
			if err := system.Points.AddBind(mount.OtherTag, console, "/dev/console", syscall.MS_BIND); err != nil {
				return err
			}
		}
	}

	for _, device := range devices {
		dev := int((device.major << 8) | (device.minor & 0xff) | ((device.minor & 0xfff00) << 12))
		path := filepath.Join(rootfsPath, device.path)
		if _, err := os.Lstat(path); os.IsNotExist(err) {
			if c.userNS {
				path = filepath.Join(c.rootfs, device.path)
				if _, err := os.Stat(device.path); os.IsNotExist(err) {
					sylog.Debugf("skipping mount, %s doesn't exists", device.path)
					continue
				}
				dirpath := filepath.Dir(path)
				if _, err := c.rpcOps.MkdirAll(dirpath, 0o755); err != nil {
					return fmt.Errorf("could not create parent directory %s: %s", dirpath, err)
				}
				if _, err := c.rpcOps.Touch(path); err != nil {
					return fmt.Errorf("could not create file %s: %s", path, err)
				}
				if err := c.rpcOps.Mount(device.path, path, "", syscall.MS_BIND, ""); err != nil {
					return fmt.Errorf("could not mount %s to %s: %s", device.path, path, err)
				}
			} else {
				dirpath := filepath.Dir(path)
				if err := os.MkdirAll(dirpath, 0o755); err != nil {
					return fmt.Errorf("could not create parent directory %s: %s", dirpath, err)
				}
				if err := syscall.Mknod(path, uint32(device.mode), dev); err != nil {
					return fmt.Errorf("could not create device %s: %s", path, err)
				}
				if device.uid != 0 || device.gid != 0 {
					if err := os.Chown(path, device.uid, device.gid); err != nil {
						return fmt.Errorf("could not change %s owner: %s", path, err)
					}
				}
			}
		}
	}

	return nil
}

func (c *container) addDevices(system *mount.System) error {
	for _, d := range c.engine.EngineConfig.OciConfig.Linux.Devices {
		var dev device

		if d.Path == "" {
			return fmt.Errorf("device path required")
		}
		dev.path = d.Path

		if d.FileMode != nil {
			dev.mode = *d.FileMode
		} else {
			dev.mode = 0o644
		}

		switch d.Type {
		case "c", "u":
			dev.mode |= syscall.S_IFCHR
			dev.major = d.Major
			dev.minor = d.Minor
		case "b":
			dev.mode |= syscall.S_IFBLK
			dev.major = d.Major
			dev.minor = d.Minor
		case "p":
			dev.mode |= syscall.S_IFIFO
		default:
			return fmt.Errorf("device type unknown for %s", d.Path)
		}

		if d.UID != nil {
			dev.uid = int(*d.UID)
		}
		if d.GID != nil {
			dev.gid = int(*d.GID)
		}

		devices = append(devices, dev)
	}

	if c.devIndex >= 0 {
		m := &c.engine.EngineConfig.OciConfig.Config.Mounts[c.devIndex]

		flags, _ := mount.ConvertOptions(m.Options)

		flags |= uintptr(syscall.MS_BIND)
		if flags&syscall.MS_RDONLY != 0 {
			if err := system.Points.AddRemount(mount.FinalTag, m.Destination, flags); err != nil {
				return err
			}
			for i := len(m.Options) - 1; i >= 0; i-- {
				if m.Options[i] == "ro" {
					m.Options = append(m.Options[:i], m.Options[i+1:]...)
					break
				}
			}
		}

		if c.engine.EngineConfig.OciConfig.Linux.Resources == nil {
			c.engine.EngineConfig.OciConfig.Linux.Resources = &specs.LinuxResources{}
		}

		// cgroupDevices are essential for operation, so must be allowed *prior* to a configured wildcard deny.
		c.engine.EngineConfig.OciConfig.Linux.Resources.Devices = append(cgroupDevices, c.engine.EngineConfig.OciConfig.Linux.Resources.Devices...)
	}

	return nil
}

func (c *container) addMaskedPathsMount(system *mount.System) error {
	paths := c.engine.EngineConfig.OciConfig.Linux.MaskedPaths

	dir, err := instance.GetDir(c.engine.CommonConfig.ContainerID, instance.OciSubDir)
	if err != nil {
		return err
	}
	nullPath := filepath.Join(dir, "null")

	if _, err := os.Stat(nullPath); os.IsNotExist(err) {
		oldmask := syscall.Umask(0)
		defer syscall.Umask(oldmask)

		if err := os.Mkdir(nullPath, 0o755); err != nil {
			return err
		}
	}

	for _, path := range paths {
		relativePath := filepath.Join(c.rootfs, path)
		rpcPath := filepath.Join(c.rpcRoot, relativePath)
		fi, err := os.Stat(rpcPath)
		if err != nil {
			sylog.Debugf("ignoring masked path %s: %s", path, err)
			continue
		}
		if fi.IsDir() {
			if err := system.Points.AddBind(mount.OtherTag, nullPath, relativePath, syscall.MS_BIND); err != nil {
				return err
			}
		} else if err := system.Points.AddBind(mount.OtherTag, "/dev/null", relativePath, syscall.MS_BIND); err != nil {
			return err
		}
	}
	return nil
}

func (c *container) addReadonlyPathsMount(system *mount.System) error {
	paths := c.engine.EngineConfig.OciConfig.Linux.ReadonlyPaths

	for _, path := range paths {
		relativePath := filepath.Join(c.rootfs, path)
		rpcPath := filepath.Join(c.rpcRoot, relativePath)
		_, err := os.Stat(rpcPath)
		if err != nil {
			sylog.Debugf("ignoring read-only path %s: %s", path, err)
			continue
		}
		if err := system.Points.AddBind(mount.OtherTag, relativePath, relativePath, syscall.MS_BIND|syscall.MS_RDONLY); err != nil {
			return err
		}
		if err := system.Points.AddRemount(mount.OtherTag, relativePath, syscall.MS_BIND|syscall.MS_RDONLY); err != nil {
			return err
		}
	}
	return nil
}

func (c *container) mount(point *mount.Point, _ *mount.System) error {
	source := point.Source
	dest := point.Destination
	flags, opts := mount.ConvertOptions(point.Options)
	optsString := strings.Join(opts, ",")
	ignore := false

	if flags&syscall.MS_REMOUNT != 0 {
		ignore = true
	}

	if !strings.HasPrefix(dest, c.rootfs) {
		rootfsPath := filepath.Join(c.rpcRoot, c.rootfs)
		relativeDest := fs.EvalRelative(dest, c.rootfs)
		procDest := filepath.Join(rootfsPath, relativeDest)

		dest = filepath.Join(c.rootfs, relativeDest)

		sylog.Debugf("Checking if %s exists", procDest)
		if _, err := os.Stat(procDest); os.IsNotExist(err) && !ignore {
			oldmask := syscall.Umask(0)
			defer syscall.Umask(oldmask)

			if point.Type != "" {
				sylog.Debugf("Creating %s", procDest)
				if c.userNS {
					if _, err := c.rpcOps.MkdirAll(dest, 0o755); err != nil {
						return err
					}
				} else {
					if err := os.MkdirAll(procDest, 0o755); err != nil {
						return err
					}
				}
			} else {
				var st syscall.Stat_t

				dir := filepath.Dir(procDest)
				if _, err := os.Stat(dir); os.IsNotExist(err) {
					sylog.Debugf("Creating parent %s", dir)
					if c.userNS {
						if err := c.rpcOps.Mkdir(filepath.Dir(dest), 0o755); err != nil {
							return err
						}
					} else {
						if err := os.MkdirAll(dir, 0o755); err != nil {
							return err
						}
					}
				}

				if err := syscall.Stat(source, &st); err != nil {
					sylog.Debugf("Ignoring %s: %s", source, err)
					return nil
				}
				switch st.Mode & syscall.S_IFMT {
				case syscall.S_IFDIR:
					sylog.Debugf("Creating dir %s", filepath.Base(procDest))
					if c.userNS {
						if err := c.rpcOps.Mkdir(dest, 0o755); err != nil {
							return err
						}
					} else {
						if err := os.Mkdir(procDest, 0o755); err != nil {
							return err
						}
					}
				case syscall.S_IFREG,
					syscall.S_IFBLK,
					syscall.S_IFCHR,
					syscall.S_IFIFO,
					syscall.S_IFSOCK:
					sylog.Debugf("Creating file %s", filepath.Base(procDest))
					if c.userNS {
						if _, err := c.rpcOps.Touch(dest); err != nil {
							return err
						}
					} else {
						if err := fs.Touch(procDest); err != nil {
							return err
						}
					}
				}
			}
		}
	} else {
		procDest := filepath.Join(c.rpcRoot, dest)

		sylog.Debugf("Checking if %s exists", procDest)
		if _, err := os.Stat(procDest); os.IsNotExist(err) {
			sylog.Warningf("destination %s doesn't exist", dest)
			return nil
		}
	}

	if ignore {
		sylog.Debugf("(re)mount %s", dest)
	} else {
		sylog.Debugf("Mount %s to %s : %s [%s]", source, dest, point.Type, optsString)
	}

	err := c.rpcOps.Mount(source, dest, point.Type, flags, optsString)
	if err != nil {
		sylog.Debugf("RPC mount error: %s", err)
	}

	return err
}
