diff options
| author | Ada Christine <adachristine18@gmail.com> | 2025-02-02 15:20:52 +0000 |
|---|---|---|
| committer | Ada Christine <adachristine18@gmail.com> | 2025-02-02 15:20:52 +0000 |
| commit | 9bc0738101979b626233ee226ece3cc8cae053ec (patch) | |
| tree | 33f608f59a3cfc062db0363c208f3b0ea05c62e3 /boot/fake_syscall.c | |
| parent | 3d806c9e2fcb6226915982b6514919a3b9dac4dc (diff) | |
full flow implemented
Diffstat (limited to 'boot/fake_syscall.c')
| -rw-r--r-- | boot/fake_syscall.c | 245 |
1 files changed, 245 insertions, 0 deletions
diff --git a/boot/fake_syscall.c b/boot/fake_syscall.c new file mode 100644 index 0000000..e2aaeca --- /dev/null +++ b/boot/fake_syscall.c @@ -0,0 +1,245 @@ +#include <kjarna/syscall.h> +#include "fake_syscall.h" +#include "kjarna.h" +#include <bits/x86_64/descriptor.h> +#include <kjarna/interface.h> +#include <posix/unistd.h> +#include <asm/x86_64/msr.h> +#include <bits/x86_64/msr.h> +#include <posix/sys/mman.h> +#include <posix/fcntl.h> + +/* + * Fake syscall mechanism - + * + * We have a GDT with entries for supervisory mode. These serve as the + * entries to satisfy the requirements of the SYSCALL instruction, as + * for some reason (possibly intentionally) the OVMF GDT is not laid out + * in a way to make use of the SYSCALL instruction possible. + * + * This causes us to have to work inside of constraints during loading time + * - All "user" mode execution entirely blocks interrupt processing. That + * means that "user" mode code must not execute "hlt", or the system will + * be locked. + * - When (if?) user input is required, it is always buffered. It is possible + * to simulate unbuffered input, at the cost of one syscall per transfer + * from "kernel" side to "user" side. This will cause high input latency. + * We will not be running Quake in this environment. + * + * These constraints are probably fine, as the loading process only needs to + * open files, map memory, etc. + */ + +void *exit_fake_syscall_stack; + +SYSV_ABI void fake_syscall_entry(void); +SYSV_ABI int64_t fake_syscall_begin(void *stack, void **return_stack_save); +SYSV_ABI int64_t fake_syscall_end(int status, void *stack); + +void *stack_alloc(void **stack_pointer, size_t alloc_size) +{ + void *block = *(char **)stack_pointer -= alloc_size; + + // maintain alignment + *(char **)stack_pointer -= alloc_size % sizeof(size_t); + + return block; +} + +struct syscall_context_return_state +{ + uint64_t rflags; + struct descriptor_table_register_long gdtr; + uint64_t return_ds; + uintptr_t return_rip; + uint64_t return_cs; +}; + +struct syscall_context +{ + uint64_t pad; + uint64_t syscall_index; + syscall_delegate_parameters parameters; + struct syscall_context_return_state state; +}; + +static struct +{ + uint16_t cs; + uint16_t ds; + uint64_t rflags; + struct descriptor_table_register_long gdtr; +} +system_context; + +static void save_system_context() +{ + __asm__ volatile ( + "movw %%cs, %0\n" + "movw %%ds, %1\n" + "pushfq\n" + "popq %3\n" + "sgdt %2\n" + : + "=m"(system_context.cs), + "=m"(system_context.ds), + "=m"(system_context.gdtr), + "=g"(system_context.rflags) + : + ); +} + +static void restore_system_context() +{ + __asm__ volatile ( + "lgdt %0\n" + "mov %w1, %%ds\n" + "mov %w1, %%es\n" + "mov %w1, %%fs\n" + "mov %w1, %%gs\n" + "mov %w1, %%gs\n" + "mov %w1, %%ss\n" + "push %q2\n" + "lea .Lflush(%%rip), %%rax\n" + "push %%rax\n" + "lretq\n" + ".Lflush:\n" + "push %q3\n" + "popfq\n" + : + : + "m"(system_context.gdtr), + "m"(system_context.ds), + "m"(system_context.cs), + "g"(system_context.rflags) + : + "rax" + ); + +} + +static struct segment_descriptor const fake_syscall_gdt[] = +{ + { 0 }, + { 0xffff, 0, 0, 0x9a, 0xaf, 0}, + { 0xffff, 0, 0, 0x92, 0xcf, 0} +}; + +static int64_t syscall_delegate_open(syscall_delegate_parameters params) +{ + return open((char *)params[0], (int)params[1], (int)params[2]); +} + +static int64_t syscall_delegate_close(syscall_delegate_parameters params) +{ + return close((int)params[0]); +} + +static int64_t syscall_delegate_lseek(syscall_delegate_parameters params) +{ + return lseek((int)params[0], (off_t)params[1], (int)params[2]); +} + +static int64_t syscall_delegate_read(syscall_delegate_parameters params) +{ + return read((int)params[0], (void *)params[1], (size_t)params[2]); +} + +static int64_t syscall_delegate_write(syscall_delegate_parameters params) +{ + return write((int)params[0], (void *)params[1], (size_t)params[2]); +} + +static int64_t syscall_delegate_mmap(syscall_delegate_parameters params) +{ + return (int64_t) mmap((void *)params[0], (size_t)params[1], (int)params[2], (int)params[3], (int)params[4], (off_t)params[5]); +} + +static int64_t syscall_delegate_munmap(syscall_delegate_parameters params) +{ + return munmap((void *)params[0], (size_t)params[1]); +} + +static int64_t syscall_delegate_exit(syscall_delegate_parameters params) +{ + fake_syscall_end((int)params[0], exit_fake_syscall_stack); + return -1; +} + +static syscall_delegate *syscall_handler_delegates[NR_SYSCALLS] = +{ + [SYS_OPEN] = syscall_delegate_open, + [SYS_CLOSE] = syscall_delegate_close, + [SYS_LSEEK] = syscall_delegate_lseek, + [SYS_READ] = syscall_delegate_read, + [SYS_WRITE] = syscall_delegate_write, + [SYS_MMAP] = syscall_delegate_mmap, + [SYS_MUNMAP] = syscall_delegate_munmap, + [SYS_EXIT] = syscall_delegate_exit +}; + +SYSV_ABI int64_t fake_syscall_handler(struct syscall_context *context) +{ + int index = (int)context->syscall_index; + if (index > NR_SYSCALLS) + { + // TODO: ENOSYS + return -1; + } + + syscall_delegate *delegate = syscall_handler_delegates[index]; + + if (delegate == nullptr) + { + // TODO: ENOSYS + return -1; + } + + restore_system_context(); + int64_t result = delegate(context->parameters); + save_system_context(); + + return result; +} + +static void install_syscall_handler(void) +{ + union msr_lstar lstar = { (uintptr_t)fake_syscall_entry }; + union msr_star star = { { 0, 1 << 3, 1 << 3 | 3 } }; + + msr_write(MSR_INDEX_LSTAR, lstar.value); + msr_write(MSR_INDEX_STAR, star.value); + + uint64_t efer = msr_read(MSR_INDEX_EFER); + efer |= 1; + msr_write(MSR_INDEX_EFER, efer); +} + +static void *create_fake_syscall_stack(void *entry_addr) +{ + constexpr size_t fake_syscall_stack_size = 0x20000; // 128KiB stack + char *stack_pointer_base = mmap(nullptr, fake_syscall_stack_size, 0, 0, -1, 0); + void *stack_pointer_head = stack_pointer_base + fake_syscall_stack_size; + + struct syscall_context *context = stack_alloc(&stack_pointer_head, sizeof(*context)); + + context->state.rflags = 0; + context->state.gdtr.base = (uintptr_t)&fake_syscall_gdt; + context->state.gdtr.limit = sizeof(fake_syscall_gdt) - 1; + context->state.return_ds = 16; + context->state.return_cs = 8; + context->state.return_rip = (uintptr_t)entry_addr; + + return stack_pointer_head; +} + +int64_t fake_syscall_start(struct kjarna_boot_image *image) +{ + install_syscall_handler(); + void *fake_stack = create_fake_syscall_stack((void *)image->entry); + save_system_context(); + int64_t result = fake_syscall_begin(fake_stack, &exit_fake_syscall_stack); + return result; + while(true); +} + |
