Skip to content

Syscall Entry Path

From the syscall instruction to kernel code — and back

x86-64 hardware mechanism

When userspace executes syscall:

  1. CPU saves: rip → rcx, rflags → r11, then clears IF
  2. CPU loads: rip from IA32_LSTAR MSR (set to entry_SYSCALL_64 at boot)
  3. CPU switches: CS/SS from IA32_STAR MSR (kernel segments)

The rax register holds the syscall number. Arguments go in: rdi, rsi, rdx, r10, r8, r9 (note: r10 not rcx — rcx is clobbered by syscall).

Return value comes back in rax. On error, the kernel negates errno and userspace glibc converts it.

entry_SYSCALL_64: the assembly entry point

/* arch/x86/entry/entry_64.S */
SYM_CODE_START(entry_SYSCALL_64)
    UNWIND_HINT_ENTRY
    ENDBR

    swapgs                  /* load kernel GS (points to per-CPU data) */
    /* Set up kernel stack using per-CPU cpu_current_top_of_stack */
    movq    %rsp, PER_CPU_VAR(cpu_tss_rw + TSS_sp2)
    SWITCH_TO_KERNEL_CR3 scratch_reg=%rsp
    movq    PER_CPU_VAR(cpu_current_top_of_stack), %rsp

    /* Push pt_regs — save all user registers */
    pushq   $__USER_DS                  /* ss */
    pushq   PER_CPU_VAR(cpu_tss_rw + TSS_sp2)  /* rsp */
    pushq   %r11                        /* rflags */
    pushq   $__USER_CS                  /* cs */
    pushq   %rcx                        /* rip */
    pushq   %rax                        /* syscall number (orig_rax) */
    PUSH_AND_CLEAR_REGS rax=$-ENOSYS    /* rax=-ENOSYS as default return */

    /* ... (IBRS mitigation, context tracking) ... */

    movq    %rsp, %rdi                  /* pt_regs pointer as first arg */
    movslq  %eax, %rsi                  /* sign-extend syscall number into second arg */
    call    do_syscall_64
    /* ... return path ... */
SYM_CODE_END(entry_SYSCALL_64)

struct pt_regs: saved register state

/* arch/x86/include/asm/ptrace.h — field names are unprefixed
 * (ax, cx, ... not rax, rcx); the register width is still 64-bit,
 * "r"-prefixing them in illustrative code is a common but wrong habit. */
struct pt_regs {
    unsigned long r15;
    unsigned long r14;
    unsigned long r13;
    unsigned long r12;
    unsigned long bp;
    unsigned long bx;
    unsigned long r11;
    unsigned long r10;
    unsigned long r9;
    unsigned long r8;
    unsigned long ax;       /* syscall number / return value */
    unsigned long cx;       /* saved rip (from syscall instruction) */
    unsigned long dx;
    unsigned long si;
    unsigned long di;
    unsigned long orig_ax;  /* original syscall number (ax before call) */
    unsigned long ip;       /* userspace instruction pointer */
    u16           cs;       /* actually a union with a FRED CS extension */
    unsigned long flags;
    unsigned long sp;       /* userspace stack pointer */
    u16           ss;       /* actually a union with a FRED SS extension */
};

do_syscall_64: the dispatch

/* arch/x86/entry/syscall_64.c */
__visible noinstr bool do_syscall_64(struct pt_regs *regs, int nr)
{
    nr = syscall_enter_from_user_mode(regs, nr);

    instrumentation_begin();
    add_random_kstack_offset();

    if (!do_syscall_x64(regs, nr) && !do_syscall_x32(regs, nr) && nr != -1) {
        /* syscall number out of range */
        regs->ax = __x64_sys_ni_syscall(regs);  /* -ENOSYS */
    }

    instrumentation_end();
    syscall_exit_to_user_mode(regs);
    /* ... SYSRET-vs-IRET decision, see below ... */
}

static __always_inline bool do_syscall_x64(struct pt_regs *regs, int nr)
{
    unsigned int unr = nr;

    if (likely(unr < NR_syscalls)) {
        unr = array_index_nospec(unr, NR_syscalls);  /* Spectre mitigation */
        regs->ax = x64_sys_call(regs, unr);           /* dispatch! */
        return true;
    }
    return false;
}

do_syscall_64() returns bool — its final lines (elided above) decide whether the fast SYSRET instruction is safe to use for the return to userspace, or whether the slower IRET path is needed.

x64_sys_call: the dispatch switch

/* arch/x86/entry/syscall_64.c */
#define __SYSCALL(nr, sym) case nr: return __x64_##sym(regs);
long x64_sys_call(const struct pt_regs *regs, unsigned int nr)
{
    switch (nr) {
    #include <asm/syscalls_64.h>
    default: return __x64_sys_ni_syscall(regs);
    }
}

A sys_call_table[] array is also built in the same file from the same syscalls_64.h, but — per the kernel's own comment above its definition — "is no longer used for system calls; kernel/trace/trace_syscalls.c still wants to know the system call address." Actual dispatch is the switch statement above.

The syscalls_64.h is generated from syscall_64.tbl:

# syscall_64.tbl (arch/x86/entry/syscalls/syscall_64.tbl)
# <number> <abi>  <name>           <entry point>
0          common  read              sys_read
1          common  write             sys_write
2          common  open              sys_open
3          common  close             sys_close
...
62         common  kill              sys_kill
...
435        common  clone3            sys_clone3

Argument passing and copying

Syscall arguments come in registers. For pointers into userspace, the kernel must copy them in — it can never directly dereference a userspace pointer:

/* Copying from userspace */
copy_from_user(kernel_buf, user_ptr, size)
    → access_ok(user_ptr, size)        /* is address in user range? */
    → __copy_from_user()               /* architecture-specific copy */
    → returns bytes NOT copied (0 = success)

/* Copying to userspace */
copy_to_user(user_ptr, kernel_buf, size)

/* Copying a string from userspace (stops at NUL or len) */
strncpy_from_user(kernel_buf, user_ptr, max_len)

/* Single value helpers (inlined, optimized) */
get_user(x, user_ptr)   /* x = *user_ptr */
put_user(x, user_ptr)   /* *user_ptr = x */

Why mandatory copies?

  • User pointers might be NULL or invalid
  • Kernel and user address spaces are separate (KPTI)
  • Userspace can change the pointed-to memory after the check (TOCTOU)

KPTI and CR3 switching

Kernel Page Table Isolation (KPTI) prevents Meltdown by using different page tables in user mode vs kernel mode:

User mode:  CR3 → user page tables (kernel not mapped, only entry stubs)
                  ↓ syscall instruction
Kernel mode: SWITCH_TO_KERNEL_CR3 (expensive TLB flush!)
             CR3 → kernel page tables (full mapping)
                  ↓ sysretq / swapgs
User mode:  SWITCH_TO_USER_CR3

On modern CPUs with PCID support, KPTI uses Process Context IDs to avoid full TLB flushes, keeping each address space's TLB entries tagged.

The vDSO: avoiding syscalls for fast operations

Some syscalls are called millions of times per second (e.g., gettimeofday, clock_gettime). The kernel exports a virtual Dynamic Shared Object (vDSO) — a small piece of code mapped into every process — that implements these without entering kernel mode:

# See the vDSO mapped in a process
cat /proc/self/maps | grep vdso
# 7fff12345000-7fff12346000 r-xp 00000000 00:00 0  [vdso]

# Functions in the vDSO
nm /proc/self/maps  # doesn't work, but:
objdump -T /lib/x86_64-linux-gnu/vdso.so.1 2>/dev/null | grep -i clock

The vDSO reads kernel timekeeping data from a shared memory page (vvar) that the kernel updates atomically. The vDSO function reads the data directly, with no privilege switch:

/* lib/vdso/gettimeofday.c — architecture-shared, not x86-specific.
 * The single struct vdso_data of older kernels is now split in two:
 * vdso_time_data (the top-level vvar page, arch clocksource state) and
 * vdso_clock (per-clock basetime/mult/shift, one instance per clock,
 * needed since time namespaces gave each namespace its own clock data). */
bool do_hres(const struct vdso_time_data *vd, const struct vdso_clock *vc,
             clockid_t clk, struct __kernel_timespec *ts)
{
    u64 sec, ns;
    u32 seq;

    do {
        /* A time-namespace'd task gets its own vdso_clock and a
         * different (still seqlock-protected) read path here. */
        if (vdso_read_begin_timens(vc, &seq))
            return do_hres_timens(vd, vc, clk, ts);

        /* TSC read + cycle-delta-to-ns math now lives in this shared
         * helper, used by every clock read path (hres, coarse, timens): */
        if (!vdso_get_timestamp(vd, vc, clk, &sec, &ns))
            return false;
    } while (vdso_read_retry(vc, seq));  /* seqlock retry */

    vdso_set_timespec(ts, sec, ns);  /* sec/ns -> tv_sec/tv_nsec, carrying ns overflow into sec */
    return true;
}

(Simplified: the real function also has an early __arch_vdso_hres_capable() bailout, and vdso_get_timestamp() itself does the TSC read and the clocksource validity check that older versions inlined directly into do_hres().)

Syscalls fast-pathed through the vDSO (no ring switch):

  • clock_gettime(CLOCK_REALTIME/MONOTONIC/...)
  • gettimeofday()
  • clock_getres()
  • getcpu()

Seccomp: restricting syscalls

The seccomp filter runs on every syscall before the dispatch table:

/* Simplified: the real chain runs through several functions —
 * syscall_enter_from_user_mode_work() (include/linux/entry-common.h)
 * checks the SYSCALL_WORK_SECCOMP bit and calls syscall_trace_enter(),
 * which reaches __seccomp_filter() in kernel/seccomp.c: */
unsigned long work = READ_ONCE(current_thread_info()->syscall_work);

if (work & SYSCALL_WORK_SECCOMP) {
    struct seccomp_data sd;
    struct seccomp_filter *match = NULL;

    populate_seccomp_data(&sd);
    u32 filter_ret = seccomp_run_filters(&sd, &match);
    u32 action = filter_ret & SECCOMP_RET_ACTION_FULL;
    /* action can be: SECCOMP_RET_ALLOW, SECCOMP_RET_KILL_(PROCESS|THREAD),
       SECCOMP_RET_ERRNO, SECCOMP_RET_TRACE, SECCOMP_RET_LOG, SECCOMP_RET_TRAP */
}

Signals and restarts

Some syscalls can be interrupted by signals (EINTR). The kernel supports automatic restart for these:

/* ERESTARTSYS: restart if no signal handler, return EINTR if one */
return -ERESTARTSYS;

/* ERESTARTNOINTR: always restart, transparent to userspace */
return -ERESTARTNOINTR;

/* ERESTARTNOHAND: restart if no handler installed */
return -ERESTARTNOHAND;

These special codes (< -MAX_ERRNO) never reach userspace — the signal path converts them to either EINTR or re-queues the syscall.

Further reading

Kernel source

Man pages

  • syscall(2) — raw syscall invocation and the x86-64 register calling convention
  • vdso(7) — the virtual dynamic shared object

LWN articles

External