Skip to main content

shadow_rs/host/
managed_thread.rs

1//! A thread of a managed process.
2//!
3//! This contains the code where the simulator can create or communicate with a managed process.
4
5use std::cell::{Cell, RefCell};
6use std::ffi::{CStr, CString};
7use std::io::Write;
8use std::os::fd::AsRawFd;
9use std::os::unix::prelude::OsStrExt;
10use std::path::PathBuf;
11use std::sync::{Arc, atomic};
12
13use linux_api::errno::Errno;
14use linux_api::posix_types::Pid;
15use linux_api::sched::CloneFlags;
16use linux_api::signal::tgkill;
17use log::{Level, debug, error, log_enabled, trace};
18use rand::RngExt as _;
19use rustix::pipe::PipeFlags;
20use rustix::process::WaitOptions;
21use shadow_shim_helper_rs::ipc::IPCData;
22use shadow_shim_helper_rs::shim_event::{
23    ShimEventAddThreadReq, ShimEventAddThreadRes, ShimEventStartRes, ShimEventSyscall,
24    ShimEventSyscallComplete, ShimEventToShadow, ShimEventToShim,
25};
26use shadow_shim_helper_rs::syscall_types::{ForeignPtr, SyscallArgs, SyscallReg};
27use shadow_shmem::allocator::ShMemBlock;
28use vasi_sync::scchannel::SelfContainedChannelError;
29
30use super::context::ThreadContext;
31use super::host::Host;
32use super::syscall::condition::SyscallCondition;
33use crate::core::worker::{WORKER_SHARED, Worker};
34use crate::cshadow;
35use crate::host::syscall::handler::SyscallHandler;
36use crate::host::syscall::types::{ForeignArrayPtr, SyscallReturn};
37use crate::utility::{VerifyPluginPathError, inject_preloads, syscall, verify_plugin_path};
38
39/// The ManagedThread's state after having been allowed to execute some code.
40#[derive(Debug)]
41#[must_use]
42pub enum ResumeResult {
43    /// Blocked on a SyscallCondition.
44    Blocked(SyscallCondition),
45    /// The native thread has exited with the given code.
46    ExitedThread(i32),
47    /// The thread's process has exited.
48    ExitedProcess,
49}
50
51pub struct ManagedThread {
52    ipc_shmem: Arc<ShMemBlock<'static, IPCData>>,
53    is_running: Cell<bool>,
54    return_code: Cell<Option<i32>>,
55
56    /* holds the event for the most recent call from the plugin/shim */
57    current_event: RefCell<ShimEventToShadow>,
58
59    native_pid: linux_api::posix_types::Pid,
60    native_tid: linux_api::posix_types::Pid,
61
62    // Value storing the current CPU affinity of the thread (more precisely,
63    // of the native thread backing this thread object). This value will be set
64    // to AFFINITY_UNINIT if CPU pinning is not enabled or if the thread has
65    // not yet been pinned to a CPU.
66    affinity: Cell<i32>,
67}
68
69impl ManagedThread {
70    pub fn native_pid(&self) -> linux_api::posix_types::Pid {
71        self.native_pid
72    }
73
74    pub fn native_tid(&self) -> linux_api::posix_types::Pid {
75        self.native_tid
76    }
77
78    /// Make the specified syscall on the native thread.
79    ///
80    /// Panics if the native thread is dead or dies during the syscall,
81    /// including if the syscall itself is SYS_exit or SYS_exit_group.
82    pub fn native_syscall(&self, ctx: &ThreadContext, n: i64, args: &[SyscallReg]) -> SyscallReg {
83        let mut syscall_args = SyscallArgs {
84            number: n,
85            args: [SyscallReg::from(0u64); 6],
86        };
87        syscall_args.args[..args.len()].copy_from_slice(args);
88        match self.continue_plugin(
89            ctx.host,
90            &ShimEventToShim::Syscall(ShimEventSyscall { syscall_args }),
91        ) {
92            ShimEventToShadow::SyscallComplete(res) => res.retval,
93            other => panic!("Unexpected response from plugin: {other:?}"),
94        }
95    }
96
97    pub fn spawn(
98        plugin_path: &CStr,
99        argv: Vec<CString>,
100        envv: Vec<CString>,
101        strace_file: Option<&std::fs::File>,
102        log_file: &std::fs::File,
103        injected_preloads: &[PathBuf],
104    ) -> Result<Self, Errno> {
105        debug!(
106            "spawning new mthread '{plugin_path:?}' with environment '{envv:?}', arguments '{argv:?}'"
107        );
108
109        let envv = inject_preloads(envv, injected_preloads);
110
111        debug!("env after preload injection: {envv:?}");
112
113        let ipc_shmem = Arc::new(shadow_shmem::allocator::shmalloc(IPCData::new()));
114
115        let child_pid =
116            Self::spawn_native(plugin_path, argv, envv, strace_file, log_file, &ipc_shmem)?;
117
118        // In Linux, the PID is equal to the TID of its first thread.
119        let native_pid = child_pid;
120        let native_tid = child_pid;
121
122        // Configure the child_pid_watcher to close the IPC channel when the child dies.
123        {
124            let worker = WORKER_SHARED.borrow();
125            let watcher = worker.as_ref().unwrap().child_pid_watcher();
126
127            watcher.register_pid(child_pid);
128            let ipc = ipc_shmem.clone();
129            watcher.register_callback(child_pid, move |_pid| {
130                ipc.from_plugin().close_writer();
131            })
132        };
133
134        trace!("waiting for start event from shim with native pid {native_pid:?}");
135        let start_req = ipc_shmem.from_plugin().receive().unwrap();
136        match &start_req {
137            ShimEventToShadow::StartReq(_) => {
138                // Expected result; shim is ready to initialize.
139            }
140            ShimEventToShadow::ProcessDeath => {
141                // The process died before initializing the shim.
142                //
143                // Reap the dead process and return an error.
144                let status =
145                    rustix::process::waitpid(Some(native_pid.into()), WaitOptions::empty())
146                        .unwrap()
147                        .unwrap();
148                if status.1.exit_status() == Some(127) {
149                    // posix_spawn(3):
150                    // > If  the child  fails  in  any  of the
151                    // > housekeeping steps described below, or fails to
152                    // > execute the desired file, it exits with a status of
153                    // > 127.
154                    debug!("posix_spawn failed to exec the process");
155                    // Assume that execve failed, and return a plausible reason
156                    // why it might have done so.
157                    // TODO: replace our usage of posix_spawn with a custom
158                    // implementation that can return the execve failure code?
159                    return Err(Errno::EPERM);
160                }
161                // TODO: handle more gracefully.
162                // * The native stdout/stderr might have a clue as to
163                // why the process died.  Consider logging a hint to
164                // check it (currently in the corresponding shimlog), or
165                // directly capture it and display it here.
166                // https://github.com/shadow/shadow/issues/3142
167                // * Consider logging a warning here and continuing on to handle
168                // the managed process exit normally. e.g. when this happens
169                // as part of an emulated `execve`, we might want to continue
170                // the simulation.
171                panic!("Child process died unexpectedly before initialization: {status:?}");
172            }
173            other => panic!("Unexpected result from shim: {other:?}"),
174        };
175
176        Ok(Self {
177            ipc_shmem,
178            is_running: Cell::new(true),
179            return_code: Cell::new(None),
180            current_event: RefCell::new(start_req),
181            native_pid,
182            native_tid,
183            affinity: Cell::new(cshadow::AFFINITY_UNINIT),
184        })
185    }
186
187    pub fn resume(
188        &self,
189        ctx: &ThreadContext,
190        syscall_handler: &mut SyscallHandler,
191    ) -> ResumeResult {
192        debug_assert!(self.is_running());
193
194        self.sync_affinity_with_worker();
195
196        loop {
197            let mut current_event = self.current_event.borrow_mut();
198            let last_event = *current_event;
199            *current_event = match last_event {
200                ShimEventToShadow::StartReq(start_req) => {
201                    // Write the serialized thread shmem handle directly to shim
202                    // memory.
203                    ctx.process
204                        .memory_borrow_mut()
205                        .write(
206                            start_req.thread_shmem_block_to_init,
207                            &ctx.thread.shmem().serialize(),
208                        )
209                        .unwrap();
210
211                    if !start_req.process_shmem_block_to_init.is_null() {
212                        // Write the serialized process shmem handle directly to
213                        // shim memory.
214                        ctx.process
215                            .memory_borrow_mut()
216                            .write(
217                                start_req.process_shmem_block_to_init,
218                                &ctx.process.shmem().serialize(),
219                            )
220                            .unwrap();
221                    }
222
223                    if !start_req.initial_working_dir_to_init.is_null() {
224                        // Write the working dir.
225                        let mut mem = ctx.process.memory_borrow_mut();
226                        let mut writer = mem.writer(ForeignArrayPtr::new(
227                            start_req.initial_working_dir_to_init,
228                            start_req.initial_working_dir_to_init_len,
229                        ));
230                        writer
231                            .write_all(ctx.process.current_working_dir().to_bytes_with_nul())
232                            .unwrap();
233                        writer.flush().unwrap();
234                    }
235
236                    // send the message to the shim to call main().
237                    trace!("sending start event code to shim");
238                    self.continue_plugin(
239                        ctx.host,
240                        &ShimEventToShim::StartRes(ShimEventStartRes {
241                            auxvec_random: ctx.host.random_mut().random(),
242                        }),
243                    )
244                }
245                ShimEventToShadow::ProcessDeath => {
246                    // The native threads are all dead or zombies. Nothing to do but
247                    // clean up.
248                    self.cleanup_after_exit_initiated();
249                    return ResumeResult::ExitedProcess;
250                }
251                ShimEventToShadow::Syscall(syscall) => {
252                    // Emulate the given syscall.
253
254                    // `exit` is tricky since it only exits the *mthread*, and we don't have a way
255                    // to be notified that the mthread has exited. We have to "fire and forget"
256                    // the command to execute the syscall natively.
257                    //
258                    // TODO: We could use a tid futex in shared memory, as set by
259                    // `set_tid_address`, to block here until the thread has
260                    // actually exited.
261                    if syscall.syscall_args.number == libc::SYS_exit {
262                        let return_code = syscall.syscall_args.args[0].into();
263                        debug!("Short-circuiting syscall exit({return_code})");
264                        self.return_code.set(Some(return_code));
265                        // Tell mthread to go ahead and make the exit syscall itself.
266                        // We *don't* call `_managedthread_continuePlugin` here,
267                        // since that'd release the ShimSharedMemHostLock, and we
268                        // aren't going to get a message back to know when it'd be
269                        // safe to take it again.
270                        self.ipc_shmem
271                            .to_plugin()
272                            .send(ShimEventToShim::SyscallDoNative);
273                        self.cleanup_after_exit_initiated();
274                        return ResumeResult::ExitedThread(return_code);
275                    }
276
277                    let scr = syscall_handler.syscall(ctx, &syscall.syscall_args).into();
278
279                    // remove the mthread's old syscall condition since it's no longer needed
280                    ctx.thread.cleanup_syscall_condition();
281
282                    assert!(self.is_running());
283
284                    match scr {
285                        SyscallReturn::Block(b) => {
286                            return ResumeResult::Blocked(unsafe {
287                                SyscallCondition::consume_from_c(b.cond)
288                            });
289                        }
290                        SyscallReturn::Done(d) => self.continue_plugin(
291                            ctx.host,
292                            &ShimEventToShim::SyscallComplete(ShimEventSyscallComplete {
293                                retval: d.retval,
294                                restartable: d.restartable,
295                            }),
296                        ),
297                        SyscallReturn::Native => {
298                            self.continue_plugin(ctx.host, &ShimEventToShim::SyscallDoNative)
299                        }
300                    }
301                }
302                ShimEventToShadow::AddThreadRes(res) => {
303                    // We get here in the child process after forking.
304
305                    // Child should have gotten 0 back from its native clone syscall.
306                    assert_eq!(res.clone_res, 0);
307
308                    // Complete the virtualized clone syscall.
309                    self.continue_plugin(
310                        ctx.host,
311                        &ShimEventToShim::SyscallComplete(ShimEventSyscallComplete {
312                            retval: 0.into(),
313                            restartable: false,
314                        }),
315                    )
316                }
317                e @ ShimEventToShadow::SyscallComplete(_) => panic!("Unexpected event: {e:?}"),
318            };
319            assert!(self.is_running());
320        }
321    }
322
323    pub fn handle_process_exit(&self) {
324        // TODO: Only do this once per process; maybe by moving into `Process`.
325        WORKER_SHARED
326            .borrow()
327            .as_ref()
328            .unwrap()
329            .child_pid_watcher()
330            .unregister_pid(self.native_pid());
331
332        self.cleanup_after_exit_initiated();
333    }
334
335    pub fn return_code(&self) -> Option<i32> {
336        self.return_code.get()
337    }
338
339    pub fn is_running(&self) -> bool {
340        self.is_running.get()
341    }
342
343    /// Execute the specified `clone` syscall in `self`, and use create a new
344    /// `ManagedThread` object to manage it. The new thread will be managed
345    /// by Shadow, and suitable for use with `Thread::wrap_mthread`.
346    ///
347    /// If the `clone` syscall fails, the native error is returned.
348    pub fn native_clone(
349        &self,
350        ctx: &ThreadContext,
351        flags: CloneFlags,
352        child_stack: ForeignPtr<()>,
353        ptid: ForeignPtr<libc::pid_t>,
354        ctid: ForeignPtr<libc::pid_t>,
355        newtls: libc::c_ulong,
356    ) -> Result<ManagedThread, linux_api::errno::Errno> {
357        let child_ipc_shmem = Arc::new(shadow_shmem::allocator::shmalloc(IPCData::new()));
358
359        // Send the IPC block for the new mthread to use.
360        let clone_res: i64 = match self.continue_plugin(
361            ctx.host,
362            &ShimEventToShim::AddThreadReq(ShimEventAddThreadReq {
363                ipc_block: child_ipc_shmem.serialize(),
364                flags: flags.bits(),
365                child_stack,
366                ptid: ptid.cast::<()>(),
367                ctid: ctid.cast::<()>(),
368                newtls,
369            }),
370        ) {
371            ShimEventToShadow::AddThreadRes(ShimEventAddThreadRes { clone_res }) => clone_res,
372            r => panic!("Unexpected result: {r:?}"),
373        };
374        let clone_res: SyscallReg = syscall::raw_return_value_to_result(clone_res)?;
375        let child_native_tid = Pid::from_raw(libc::pid_t::from(clone_res)).unwrap();
376        trace!("native clone treated tid {child_native_tid:?}");
377
378        trace!("waiting for start event from shim with native tid {child_native_tid:?}");
379        let start_req = child_ipc_shmem.from_plugin().receive().unwrap();
380        match &start_req {
381            ShimEventToShadow::StartReq(_) => (),
382            other => panic!("Unexpected result from shim: {other:?}"),
383        };
384
385        let native_pid = if flags.contains(CloneFlags::CLONE_THREAD) {
386            self.native_pid
387        } else {
388            child_native_tid
389        };
390
391        if !flags.contains(CloneFlags::CLONE_THREAD) {
392            // Child is a new process; register it.
393            WORKER_SHARED
394                .borrow()
395                .as_ref()
396                .unwrap()
397                .child_pid_watcher()
398                .register_pid(native_pid);
399        }
400
401        // Register the child thread's IPC block with the ChildPidWatcher.
402        {
403            let child_ipc_shmem = child_ipc_shmem.clone();
404            WORKER_SHARED
405                .borrow()
406                .as_ref()
407                .unwrap()
408                .child_pid_watcher()
409                .register_callback(native_pid, move |_pid| {
410                    child_ipc_shmem.from_plugin().close_writer();
411                })
412        };
413
414        Ok(Self {
415            ipc_shmem: child_ipc_shmem,
416            is_running: Cell::new(true),
417            return_code: Cell::new(None),
418            current_event: RefCell::new(start_req),
419            native_pid,
420            native_tid: child_native_tid,
421            // TODO: can we assume it's inherited from the current thread affinity?
422            affinity: Cell::new(cshadow::AFFINITY_UNINIT),
423        })
424    }
425
426    #[must_use]
427    fn continue_plugin(&self, host: &Host, event: &ShimEventToShim) -> ShimEventToShadow {
428        // Update shared state before transferring control.
429        host.shim_shmem_lock_borrow_mut().unwrap().max_runahead_time =
430            Worker::max_event_runahead_time(host);
431        host.shim_shmem()
432            .sim_time
433            .store(Worker::current_time().unwrap(), atomic::Ordering::Relaxed);
434
435        // Release lock so that plugin can take it. Reacquired in `wait_for_next_event`.
436        host.unlock_shmem();
437
438        self.ipc_shmem.to_plugin().send(*event);
439
440        let event = match self.ipc_shmem.from_plugin().receive() {
441            Ok(e) => e,
442            Err(SelfContainedChannelError::WriterIsClosed) => ShimEventToShadow::ProcessDeath,
443        };
444
445        // Reacquire the shared memory lock, now that the shim has yielded control
446        // back to us.
447        host.lock_shmem();
448
449        // Update time, which may have been incremented in the shim.
450        let shim_time = host.shim_shmem().sim_time.load(atomic::Ordering::Relaxed);
451        if log_enabled!(Level::Trace) {
452            let worker_time = Worker::current_time().unwrap();
453            if shim_time != worker_time {
454                trace!(
455                    "Updating time from {worker_time:?} to {shim_time:?} (+{:?})",
456                    shim_time - worker_time
457                );
458            }
459        }
460        Worker::set_current_time(shim_time);
461
462        event
463    }
464
465    /// To be called after we expect the native thread to have exited, or to
466    /// exit imminently.
467    fn cleanup_after_exit_initiated(&self) {
468        if !self.is_running.get() {
469            return;
470        }
471        self.wait_for_native_exit();
472        trace!("child {:?} exited", self.native_tid());
473        self.is_running.set(false);
474    }
475
476    /// Wait until the managed thread is no longer running.
477    fn wait_for_native_exit(&self) {
478        let native_pid = self.native_pid();
479        let native_tid = self.native_tid();
480
481        // We use `tgkill` and `/proc/x/stat` to detect whether the thread is still running,
482        // looping until it doesn't.
483        //
484        // Alternatively we could use `set_tid_address` or `set_robust_list` to
485        // be notified on a futex. Those are a bit underdocumented and fragile,
486        // though. In practice this shouldn't have to loop significantly.
487        trace!("Waiting for native thread {native_pid:?}.{native_tid:?} to exit");
488        loop {
489            if self.ipc_shmem.from_plugin().writer_is_closed() {
490                // This indicates that the whole process has stopped executing;
491                // no need to poll the individual thread.
492                break;
493            }
494            match tgkill(native_pid, native_tid, None) {
495                Err(Errno::ESRCH) => {
496                    trace!("Thread is done exiting; proceeding with cleanup");
497                    break;
498                }
499                Err(e) => {
500                    error!("Unexpected tgkill error: {e:?}");
501                    break;
502                }
503                Ok(()) if native_pid == native_tid => {
504                    // Thread leader could be in a zombie state waiting for
505                    // the other threads to exit.
506                    let filename = format!("/proc/{}/stat", native_pid.as_raw_nonzero().get());
507                    let stat = match std::fs::read_to_string(filename) {
508                        Err(e) => {
509                            assert!(e.kind() == std::io::ErrorKind::NotFound);
510                            trace!("tgl {native_pid:?} is fully dead");
511                            break;
512                        }
513                        Ok(s) => s,
514                    };
515                    if stat.contains(") Z") {
516                        trace!("tgl {native_pid:?} is a zombie");
517                        break;
518                    }
519                    // Still alive and in a non-zombie state; continue
520                }
521                Ok(()) => {
522                    // Thread is still alive; continue.
523                }
524            };
525            std::thread::yield_now();
526        }
527    }
528
529    fn sync_affinity_with_worker(&self) {
530        let current_affinity = scheduler::core_affinity()
531            .map(|x| i32::try_from(x).unwrap())
532            .unwrap_or(cshadow::AFFINITY_UNINIT);
533        self.affinity.set(unsafe {
534            cshadow::affinity_setProcessAffinity(
535                self.native_tid().as_raw_nonzero().get(),
536                current_affinity,
537                self.affinity.get(),
538            )
539        });
540    }
541
542    fn spawn_native(
543        plugin_path: &CStr,
544        argv: Vec<CString>,
545        envv: Vec<CString>,
546        strace_file: Option<&std::fs::File>,
547        shimlog_file: &std::fs::File,
548        shmem_block: &ShMemBlock<IPCData>,
549    ) -> Result<Pid, Errno> {
550        // Preemptively check for likely reasons that execve might fail.
551        // In particular we want to ensure that we  don't launch a statically
552        // linked executable, since we'd then deadlock the whole simulation
553        // waiting for the plugin to initialize.
554        //
555        // This is also helpful since we can't retrieve specific `execve` errors
556        // through `posix_spawn`.
557        fn map_verify_err(e: VerifyPluginPathError) -> Errno {
558            match e {
559                // execve(2): ENOENT The file pathname [...] does not exist.
560                VerifyPluginPathError::NotFound => Errno::ENOENT,
561                // execve(2): EACCES The file or a script interpreter is not a regular file.
562                VerifyPluginPathError::NotFile => Errno::EACCES,
563                // execve(2): EACCES Execute permission is denied for the file or a script or ELF interpreter.
564                VerifyPluginPathError::NotExecutable => Errno::EACCES,
565                // execve(2): ENOEXEC An executable is not in a recognized
566                // format, is for the wrong architecture, or has some other
567                // format error that means it cannot be executed.
568                VerifyPluginPathError::UnknownFileType => Errno::ENOEXEC,
569                VerifyPluginPathError::NotDynamicallyLinkedElf => Errno::ENOEXEC,
570                VerifyPluginPathError::IncompatibleInterpreter(e) => map_verify_err(*e),
571                // execve(2): EACCES Search permission is denied on a component
572                // of the path prefix of pathname or the name of a script
573                // interpreter.
574                VerifyPluginPathError::PathPermissionDenied => Errno::EACCES,
575                VerifyPluginPathError::UnhandledIoError(_) => {
576                    // Arbitrary error that should be handled by callers.
577                    Errno::ENOEXEC
578                }
579            }
580        }
581        verify_plugin_path(std::ffi::OsStr::from_bytes(plugin_path.to_bytes()))
582            .map_err(map_verify_err)?;
583
584        // posix_spawn is documented as taking pointers to *mutable* char for argv and
585        // envv. It *probably* doesn't actually mutate them, but we
586        // conservatively give it what it asks for. We have to "reconstitute"
587        // the CString's after the fork + exec to deallocate them.
588        let argv_ptrs: Vec<*mut i8> = argv
589            .into_iter()
590            .map(CString::into_raw)
591            // the last element of argv must be NULL
592            .chain(std::iter::once(std::ptr::null_mut()))
593            .collect();
594        let envv_ptrs: Vec<*mut i8> = envv
595            .into_iter()
596            .map(CString::into_raw)
597            // the last element of argv must be NULL
598            .chain(std::iter::once(std::ptr::null_mut()))
599            .collect();
600
601        let mut file_actions: libc::posix_spawn_file_actions_t = shadow_pod::zeroed();
602        Errno::result_from_libc_errnum(unsafe {
603            libc::posix_spawn_file_actions_init(&mut file_actions)
604        })
605        .unwrap();
606
607        // Set up stdin
608        let (stdin_reader, stdin_writer) = rustix::pipe::pipe_with(PipeFlags::CLOEXEC).unwrap();
609        Errno::result_from_libc_errnum(unsafe {
610            libc::posix_spawn_file_actions_adddup2(
611                &mut file_actions,
612                stdin_reader.as_raw_fd(),
613                libc::STDIN_FILENO,
614            )
615        })
616        .unwrap();
617
618        // Dup straceFd; the dup'd descriptor won't have O_CLOEXEC set.
619        //
620        // Since dup2 is a no-op when the new and old file descriptors are equal, we have
621        // to arrange to call dup2 twice - first to a temporary descriptor, and then back
622        // to the original descriptor number.
623        //
624        // Here we use STDOUT_FILENO as the temporary descriptor, since we later
625        // replace that below.
626        //
627        // Once we drop support for platforms with glibc older than 2.29, we *could*
628        // consider taking advantage of a new feature that would let us just use a
629        // single `posix_spawn_file_actions_adddup2` call with equal descriptors.
630        // OTOH it's a non-standard extension, and I think ultimately uses the same
631        // number of syscalls, so it might be better to continue using this slightly
632        // more awkward method anyway.
633        // https://github.com/bminor/glibc/commit/805334b26c7e6e83557234f2008497c72176a6cd
634        // https://austingroupbugs.net/view.php?id=411
635        if let Some(strace_file) = strace_file {
636            Errno::result_from_libc_errnum(unsafe {
637                libc::posix_spawn_file_actions_adddup2(
638                    &mut file_actions,
639                    strace_file.as_raw_fd(),
640                    libc::STDOUT_FILENO,
641                )
642            })
643            .unwrap();
644            Errno::result_from_libc_errnum(unsafe {
645                libc::posix_spawn_file_actions_adddup2(
646                    &mut file_actions,
647                    libc::STDOUT_FILENO,
648                    strace_file.as_raw_fd(),
649                )
650            })
651            .unwrap();
652        }
653
654        // set stdout/stderr as the shim log. This also clears the FD_CLOEXEC flag.
655        Errno::result_from_libc_errnum(unsafe {
656            libc::posix_spawn_file_actions_adddup2(
657                &mut file_actions,
658                shimlog_file.as_raw_fd(),
659                libc::STDOUT_FILENO,
660            )
661        })
662        .unwrap();
663        Errno::result_from_libc_errnum(unsafe {
664            libc::posix_spawn_file_actions_adddup2(
665                &mut file_actions,
666                shimlog_file.as_raw_fd(),
667                libc::STDERR_FILENO,
668            )
669        })
670        .unwrap();
671
672        let mut spawn_attr: libc::posix_spawnattr_t = shadow_pod::zeroed();
673        Errno::result_from_libc_errnum(unsafe { libc::posix_spawnattr_init(&mut spawn_attr) })
674            .unwrap();
675
676        // In versions of glibc before 2.24, we need this to tell posix_spawn
677        // to use vfork instead of fork. In later versions it's a no-op.
678        Errno::result_from_libc_errnum(unsafe {
679            libc::posix_spawnattr_setflags(
680                &mut spawn_attr,
681                libc::POSIX_SPAWN_USEVFORK.try_into().unwrap(),
682            )
683        })
684        .unwrap();
685
686        let child_pid_res = {
687            let mut child_pid = -1;
688            Errno::result_from_libc_errnum(unsafe {
689                libc::posix_spawn(
690                    &mut child_pid,
691                    plugin_path.as_ptr(),
692                    &file_actions,
693                    &spawn_attr,
694                    argv_ptrs.as_ptr(),
695                    envv_ptrs.as_ptr(),
696                )
697            })
698            .map(|_| Pid::from_raw(child_pid).unwrap_or_else(|| panic!("Invalid pid: {child_pid}")))
699        };
700
701        // Write the serialized shmem descriptor to the stdin pipe. The pipe
702        // buffer should be large enough that we can write it all without having
703        // to wait for data to be read.
704        if child_pid_res.is_ok() {
705            // we avoid using the rustix write wrapper here, since we can't guarantee
706            // that all bytes of the serialized shmem block are initd, and hence
707            // can't safely construct the &[u8] that it wants.
708            let serialized = shmem_block.serialize();
709            let serialized_bytes = shadow_pod::as_u8_slice(&serialized);
710            let written = Errno::result_from_libc_errno(-1, unsafe {
711                libc::write(
712                    stdin_writer.as_raw_fd(),
713                    serialized_bytes.as_ptr().cast(),
714                    serialized_bytes.len(),
715                )
716            })
717            .unwrap();
718            // TODO: loop if needed. Shouldn't be in practice, though.
719            assert_eq!(written, isize::try_from(serialized_bytes.len()).unwrap());
720        }
721
722        Errno::result_from_libc_errnum(unsafe {
723            libc::posix_spawn_file_actions_destroy(&mut file_actions)
724        })
725        .unwrap();
726        Errno::result_from_libc_errnum(unsafe { libc::posix_spawnattr_destroy(&mut spawn_attr) })
727            .unwrap();
728
729        // Drop the cloned argv and env.
730        drop(
731            argv_ptrs
732                .into_iter()
733                .filter(|p| !p.is_null())
734                .map(|p| unsafe { CString::from_raw(p) }),
735        );
736        drop(
737            envv_ptrs
738                .into_iter()
739                .filter(|p| !p.is_null())
740                .map(|p| unsafe { CString::from_raw(p) }),
741        );
742
743        debug!(
744            "starting process {}, result: {child_pid_res:?}",
745            plugin_path.to_str().unwrap()
746        );
747
748        child_pid_res
749    }
750
751    /// `ManagedThread` panics if dropped while the underlying process is still running,
752    /// since otherwise that process could continue writing to shared memory regions
753    /// that shadow reallocates.
754    ///
755    /// This method kills the process that `self` belongs to (not just the
756    /// thread!) and then drops `self`.
757    pub fn kill_and_drop(self) {
758        if let Err(err) =
759            rustix::process::kill_process(self.native_pid().into(), rustix::process::Signal::KILL)
760        {
761            log::warn!(
762                "Couldn't kill managed process {:?}. kill: {:?}",
763                self.native_pid(),
764                err
765            );
766        }
767        self.handle_process_exit();
768    }
769}
770
771impl Drop for ManagedThread {
772    fn drop(&mut self) {
773        // Dropping while the thread is running is unsound because the running
774        // thread still has access to shared memory regions that will be
775        // deallocated, and potentially reallocated for another purpose. The
776        // running thread accessing a deallocated or repurposed memory region
777        // can cause numerous problems.
778        assert!(!self.is_running());
779    }
780}