arch/x86_64: Implement up_addrenv_fork() and provide POSIX fork().

Duplicate an address environment into freshly allocated pages mapped at the
same virtual addresses, which is what POSIX fork() is built on.

x86_64_fork_syscall() then lets the child run at the parent's stack addresses.
A pointer to a stack local taken before fork() must name the same object in
the child that it named in the parent, so the child adopts the parent's stack
geometry rather than being given a relocated copy; the parent's stack is
already in the duplicate, at the parent's address, with its contents.  That
shows up as a zero offset, which also means the copy would have the same
source and destination, so both the copy and the frame-pointer relocation are
skipped.

Build-verified on qemu-intel64:knsh_romfs.  NuttX on qemu-intel64 requires
tsc-deadline and pcid, which TCG does not implement, so it cannot be run on
this host.

Assisted-by: Claude Code:claude-opus-5
Signed-off-by: Marco Casaroli <marco.casaroli@gmail.com>
This commit is contained in:
Marco Casaroli 2026-08-09 15:43:18 +02:00 • committed by Alan C. Assis
parent 026f77d940
commit 95c326704e
4 changed files with 209 additions and 6 deletions

View file

@ -211,7 +211,7 @@ Known gaps
complete -- ``addrenv_fork()``, the ``up_addrenv_fork()`` hook, the syscall, the
libc wrapper and the ``ostest`` case -- so an architecture provides ``fork()``
by implementing ``up_addrenv_fork()`` and selecting ``CONFIG_ARCH_HAVE_FORK``,
with no further generic work.
with no further generic work. x86_64 selects it today.
**A windowed ABI needs its stack rebased, not just copied.** On Xtensa,
giving a child a relocated copy of the parent's stack takes more than the copy:

View file

@ -493,6 +493,7 @@ config ARCH_HAVE_VFORK
config ARCH_HAVE_FORK
bool
default y if ARCH_X86_64 && !BUILD_PROTECTED
default n
depends on ARCH_ADDRENV
---help---

View file

@ -483,6 +483,190 @@ errout:
return ret;
}
#ifdef CONFIG_ARCH_HAVE_FORK
/****************************************************************************
* Name: vaddr_is_text
*
* Description:
* Check if a vaddr is part of the .text area, which is mapped read/execute
* while everything else is mapped read/write. The two arms mirror exactly
* the two layouts up_addrenv_create() builds.
*
****************************************************************************/
static inline bool vaddr_is_text(const arch_addrenv_t *addrenv,
uintptr_t vaddr)
{
#if (CONFIG_ARCH_TEXT_VBASE != 0x0) && (CONFIG_ARCH_HEAP_VBASE != 0x0)
UNUSED(addrenv);
return vaddr >= CONFIG_ARCH_TEXT_VBASE && vaddr < ARCH_TEXT_VEND;
#else
/* Contiguous layout: the reserve sits below .text, and .data begins where
* .text ends.
*/
return vaddr >= addrenv->textvbase && vaddr < addrenv->datavbase;
#endif
}
/****************************************************************************
* Name: up_addrenv_fork
*
* Description:
* Duplicate an address environment for POSIX fork(). The destination gets
* its own page tables and its own physical pages, holding a copy of the
* source's contents and mapped at the same virtual addresses.
*
* The walk mirrors up_addrenv_destroy(): every page table (PT) the source
* has under its page directory is visited, and every page it maps is
* duplicated. The copy is eager -- there is no copy-on-write -- so this
* needs as much free page memory as the parent occupies.
*
* Input Parameters:
* src - The address environment to be duplicated.
* dest - The location to receive the duplicate.
*
* Returned Value:
* Zero (OK) on success; a negated errno value on failure.
*
****************************************************************************/
int up_addrenv_fork(const arch_addrenv_t *src, arch_addrenv_t *dest)
{
uintptr_t *sptprev;
uintptr_t *sptlast;
uintptr_t dptprev;
uintptr_t *dptlast;
uintptr_t paddr;
uintptr_t vaddr;
uintptr_t pgvaddr;
size_t pdsize;
int i;
int j;
int ret;
DEBUGASSERT(src && dest);
memset(dest, 0, sizeof(arch_addrenv_t));
/* Give the child its own static page tables and kernel mappings */
ret = create_spgtables(dest);
if (ret < 0)
{
berr("ERROR: Failed to create static page tables\n");
goto errout;
}
copy_kernel_mappings(dest);
/* The duplicate lives at the same virtual addresses as the original */
dest->textvbase = src->textvbase;
dest->datavbase = src->datavbase;
dest->heapvbase = src->heapvbase;
dest->heapsize = src->heapsize;
dest->cr3 = mmu_cr3_reg(dest->spgtables[0], 0);
/* Make sure the source's page tables are visible before walking them */
UP_DSB();
UP_DMB();
/* Each entry of the page directory -- the last static level -- covers one
* page table's worth of address space.
*/
vaddr = ARCH_ADDRENV_VBASE;
pdsize = (size_t)X86_MMU_ENTRIES_PER_PGT * X86_MMU_PAGE_SIZE;
sptprev = (uintptr_t *)x86_64_pgvaddr(src->spgtables[ARCH_SPGTS - 1]);
dptprev = x86_64_pgvaddr(dest->spgtables[ARCH_SPGTS - 1]);
if (sptprev == NULL || dptprev == 0)
{
ret = -EINVAL;
goto errout;
}
for (i = 0; i < X86_MMU_ENTRIES_PER_PGT; i++, vaddr += pdsize)
{
sptlast = (uintptr_t *)x86_64_pgvaddr(mmu_pte_to_paddr(sptprev[i]));
if (sptlast == NULL)
{
continue;
}
/* Hook the static tables up for this address, then give the child its
* own page table here.
*/
map_spgtables(dest, vaddr);
paddr = mm_pgalloc(1);
if (!paddr)
{
ret = -ENOMEM;
goto errout;
}
x86_64_pgwipe(paddr);
mmu_ln_setentry(ARCH_SPGTS - 1, dptprev, paddr, vaddr, MMU_UPGT_FLAGS);
dptlast = (uintptr_t *)x86_64_pgvaddr(paddr);
for (j = 0; j < X86_MMU_ENTRIES_PER_PGT; j++)
{
uintptr_t srcpage;
uintptr_t destpage;
srcpage = mmu_pte_to_paddr(sptlast[j]);
if (!srcpage)
{
continue;
}
pgvaddr = vaddr + (uintptr_t)j * X86_MMU_PAGE_SIZE;
if (vaddr_is_shm(pgvaddr))
{
/* Shared memory stays shared across fork(). Map the very same
* page; up_addrenv_destroy() knows not to free SHM pages.
*/
dptlast[j] = sptlast[j];
continue;
}
destpage = mm_pgalloc(1);
if (!destpage)
{
ret = -ENOMEM;
goto errout;
}
memcpy((void *)x86_64_pgvaddr(destpage),
(const void *)x86_64_pgvaddr(srcpage), X86_MMU_PAGE_SIZE);
mmu_ln_setentry(ARCH_SPGTS, (uintptr_t)dptlast, destpage, pgvaddr,
vaddr_is_text(src, pgvaddr) ? MMU_UTEXT_FLAGS
: MMU_UDATA_FLAGS);
}
}
UP_DSB();
UP_DMB();
#ifdef CONFIG_SMP
x86_64_tlb_shootdown();
#endif
return OK;
errout:
up_addrenv_destroy(dest);
return ret;
}
#endif /* CONFIG_ARCH_HAVE_FORK */
/****************************************************************************
* Name: up_addrenv_destroy
*

View file

@ -399,7 +399,17 @@ static pid_t x86_64_fork_syscall(bool vfork, struct tcb_s *parent)
newsp = newtop - stackutil;
offset = newtop - stacktop;
memcpy((void *)newsp, (const void *)rsp, stackutil);
/* A zero offset means the child is running on the parent's stack
* addresses: a fork() child, which inherited them and whose duplicated
* address environment already holds a copy of the contents. There is then
* nothing to copy -- source and destination would be the same region --
* and nothing to relocate.
*/
if (offset != 0)
{
memcpy((void *)newsp, (const void *)rsp, stackutil);
}
sinfo("Old stack top:%08" PRIx64 " RSP:%08" PRIx64 "\n", stacktop, rsp);
sinfo("New stack top:%08" PRIx64 " RSP:%08" PRIx64 "\n", newtop, newsp);
@ -430,11 +440,19 @@ static pid_t x86_64_fork_syscall(bool vfork, struct tcb_s *parent)
child->xcp.regs[REG_RSI] = sregs[REG_RSI];
child->xcp.regs[REG_RDI] = sregs[REG_RDI];
/* The frame pointer moves with the stack it points into */
/* The frame pointer moves with the stack it points into. With a zero
* offset it does not move at all, and neither does the saved chain it
* heads: see the note above the copy.
*/
child->xcp.regs[REG_RBP] = x86_64_fork_reloc(sregs[REG_RBP], rsp,
stacktop, offset);
x86_64_fork_relocfp(sregs[REG_RBP], rsp, stacktop, offset);
child->xcp.regs[REG_RBP] = sregs[REG_RBP];
if (offset != 0)
{
child->xcp.regs[REG_RBP] = x86_64_fork_reloc(sregs[REG_RBP], rsp,
stacktop, offset);
x86_64_fork_relocfp(sregs[REG_RBP], rsp, stacktop, offset);
}
/* Build the interrupt frame the child is resumed from. RIP and RFLAGS
* come out of RCX and R11, and the selectors are the ones SYSRETQ derives