Skip to main content

shadow_rs/host/syscall/handler/
sched.rs

1use std::mem::MaybeUninit;
2
3use bitflags::Flags;
4use linux_api::errno::Errno;
5use linux_api::posix_types::kernel_pid_t;
6use linux_api::rseq::{rseq, rseq_flags};
7use linux_api::sched::{SCHED_RESET_ON_FORK, Sched, sched_attr};
8use log::warn;
9use shadow_shim_helper_rs::explicit_drop::{ExplicitDrop, ExplicitDropper};
10use shadow_shim_helper_rs::syscall_types::ForeignPtr;
11
12use crate::host::syscall::handler::{SyscallContext, SyscallHandler};
13use crate::host::syscall::type_formatting::SyscallNonDeterministicArg;
14use crate::host::syscall::types::ForeignArrayPtr;
15use crate::host::thread::{Thread, ThreadId};
16
17// We always report that the thread is running on CPU 0, Node 0
18const CURRENT_CPU: u32 = 0;
19
20/// Run `f` on the thread targeted by the scheduler syscalls.
21///
22/// Linux treats a tid of 0 as the current thread for these calls. For any other tid, borrow the
23/// emulated thread through a cloned `RootedRc` and explicitly drop that clone before returning.
24fn with_sched_target_thread<T>(
25    ctx: &SyscallContext,
26    tid: kernel_pid_t,
27    f: impl FnOnce(&Thread) -> T,
28) -> Result<T, Errno> {
29    let current_tid = kernel_pid_t::from(ctx.objs.thread.id());
30    if tid == 0 || tid == current_tid {
31        return Ok(f(ctx.objs.thread));
32    }
33
34    let target_tid = ThreadId::try_from(tid).or(Err(Errno::ESRCH))?;
35    let Some(thread_rc) = ctx.objs.host.thread_cloned_rc(target_tid) else {
36        return Err(Errno::ESRCH);
37    };
38    let thread_rc =
39        ExplicitDropper::new(thread_rc, |value| value.explicit_drop(ctx.objs.host.root()));
40    let thread = thread_rc.borrow(ctx.objs.host.root());
41    Ok(f(&thread))
42}
43
44/// Validate the priority rules for the scheduler policies Shadow tracks.
45///
46/// Non-realtime policies use a fixed priority of 0, while Linux accepts priorities 1 through 99
47/// for `SCHED_FIFO` and `SCHED_RR`.
48fn validate_sched_attrs(policy: Sched, priority: std::ffi::c_int) -> Result<(), Errno> {
49    match policy {
50        Sched::SCHED_NORMAL | Sched::SCHED_BATCH | Sched::SCHED_IDLE if priority == 0 => Ok(()),
51        Sched::SCHED_FIFO | Sched::SCHED_RR if (1..=99).contains(&priority) => Ok(()),
52        _ => Err(Errno::EINVAL),
53    }
54}
55
56// Our simulated cpu set size.
57const KERNEL_CPUSETSIZE_BYTES: usize = 1;
58
59impl SyscallHandler {
60    log_syscall!(
61        sched_getaffinity,
62        /* rv */ i32,
63        // Non-deterministic due to https://github.com/shadow/shadow/issues/3626
64        /* pid */
65        SyscallNonDeterministicArg<kernel_pid_t>,
66        /* cpusetsize */ usize,
67        /* mask */ *const std::ffi::c_void,
68    );
69    pub fn sched_getaffinity(
70        ctx: &mut SyscallContext,
71        tid: kernel_pid_t,
72        cpusetsize: usize,
73        // sched_getaffinity(2):
74        // > The underlying system calls (which represent CPU masks as bit masks
75        // > of type unsigned long *) impose no restriction on the size of the CPU
76        // > mask
77        mask_ptr: ForeignPtr<std::ffi::c_ulong>,
78    ) -> Result<std::ffi::c_int, Errno> {
79        // Check that `tid` is valid.
80        let tid = ThreadId::try_from(tid).or(Err(Errno::ESRCH))?;
81        if !ctx.objs.host.has_thread(tid) && kernel_pid_t::from(tid) != 0 {
82            return Err(Errno::ESRCH);
83        }
84        // We currently always write back a mask that says the thread can
85        // execute only on CPU 0.
86
87        if cpusetsize < KERNEL_CPUSETSIZE_BYTES {
88            // sched_getaffinity(2): EINVAL (sched_getaffinity() and, in kernels
89            // before 2.6.9, sched_setaffinity()) cpusetsize is smaller than the
90            // size of the affinity mask used by the kernel.
91            return Err(Errno::EINVAL);
92        }
93        let mask_ptr = ForeignArrayPtr::new(mask_ptr.cast::<u8>(), KERNEL_CPUSETSIZE_BYTES);
94        // Our simulated mask says we can run only on CPU 0, which is 1 in the
95        // first byte, assuming little endian.
96        let mask: [u8; KERNEL_CPUSETSIZE_BYTES] = [1u8];
97
98        ctx.objs
99            .process
100            .memory_borrow_mut()
101            .copy_to_ptr(mask_ptr, &mask)?;
102
103        Ok(KERNEL_CPUSETSIZE_BYTES.try_into().unwrap())
104    }
105
106    log_syscall!(
107        sched_setaffinity,
108        /* rv */ i32,
109        /* pid */ kernel_pid_t,
110        /* cpusetsize */ usize,
111        /* mask */ *const std::ffi::c_void,
112    );
113    pub fn sched_setaffinity(
114        ctx: &mut SyscallContext,
115        tid: kernel_pid_t,
116        cpusetsize: usize,
117        // sched_getaffinity(2):
118        // > The underlying system calls (which represent CPU masks as bit masks
119        // > of type unsigned long *) impose no restriction on the size of the CPU
120        // > mask
121        mask_ptr: ForeignPtr<std::ffi::c_ulong>,
122    ) -> Result<(), Errno> {
123        let tid = ThreadId::try_from(tid).or(Err(Errno::ESRCH))?;
124        if !ctx.objs.host.has_thread(tid) && kernel_pid_t::from(tid) != 0 {
125            return Err(Errno::ESRCH);
126        };
127
128        // Shadow doesn't have users, so no need to check for permissions
129
130        let mut mask = [0u8; KERNEL_CPUSETSIZE_BYTES];
131        let mask_ptr = mask_ptr.cast::<u8>();
132        let mask_ptr = ForeignArrayPtr::new(mask_ptr, cpusetsize);
133        let to_copy = std::cmp::min(mask.len(), mask_ptr.len());
134        ctx.objs
135            .process
136            .memory_borrow()
137            .copy_from_ptr(&mut mask[..to_copy], mask_ptr.slice(..to_copy))?;
138
139        // We simulate that
140        // this assumes little endian
141        if mask[0] & 0x01 == 0 {
142            // "the  affinity  bit  mask  mask  contains  no  processors  that
143            // are currently physically on the system and permitted to the
144            // thread"
145            return Err(Errno::EINVAL);
146        }
147
148        // We currently throw away the mask. TODO: store it, and return it
149        // in sched_getaffinity.
150
151        Ok(())
152    }
153
154    log_syscall!(
155        sched_getparam,
156        /* rv */ i32,
157        /* pid */ kernel_pid_t,
158        /* param */ *const linux_api::sched::sched_attr,
159    );
160    pub fn sched_getparam(
161        ctx: &mut SyscallContext,
162        tid: kernel_pid_t,
163        param_ptr: ForeignPtr<sched_attr>,
164    ) -> Result<(), Errno> {
165        warn_once_then_debug!(
166            "sched_getparam() only returns tracked scheduler state; Shadow does not emulate Linux scheduling behavior"
167        );
168
169        let priority = with_sched_target_thread(ctx, tid, |thread| thread.sched_priority())?;
170        ctx.objs
171            .process
172            .memory_borrow_mut()
173            .write(param_ptr.cast::<std::ffi::c_int>(), &priority)?;
174
175        Ok(())
176    }
177
178    log_syscall!(
179        sched_getscheduler,
180        /* rv */ i32,
181        /* pid */ kernel_pid_t,
182    );
183    pub fn sched_getscheduler(
184        ctx: &mut SyscallContext,
185        tid: kernel_pid_t,
186    ) -> Result<std::ffi::c_int, Errno> {
187        warn_once_then_debug!(
188            "sched_getscheduler() only returns tracked scheduler state; Shadow does not emulate Linux scheduling behavior"
189        );
190
191        with_sched_target_thread(ctx, tid, |thread| {
192            let mut policy = i32::from(thread.sched_policy());
193            if thread.sched_reset_on_fork() {
194                policy |= SCHED_RESET_ON_FORK;
195            }
196            policy
197        })
198    }
199
200    log_syscall!(
201        sched_setparam,
202        /* rv */ i32,
203        /* pid */ kernel_pid_t,
204        /* param */ *const linux_api::sched::sched_attr,
205    );
206    pub fn sched_setparam(
207        ctx: &mut SyscallContext,
208        tid: kernel_pid_t,
209        param_ptr: ForeignPtr<sched_attr>,
210    ) -> Result<(), Errno> {
211        warn_once_then_debug!(
212            "sched_setparam() only updates tracked scheduler state; Shadow does not emulate Linux scheduling behavior"
213        );
214
215        let new_priority = ctx
216            .objs
217            .process
218            .memory_borrow()
219            .read(param_ptr.cast::<std::ffi::c_int>())?;
220        let (policy, reset_on_fork) = with_sched_target_thread(ctx, tid, |thread| {
221            (thread.sched_policy(), thread.sched_reset_on_fork())
222        })?;
223
224        validate_sched_attrs(policy, new_priority)?;
225
226        with_sched_target_thread(ctx, tid, |thread| {
227            thread.set_sched_attrs(policy, reset_on_fork, new_priority);
228        })?;
229
230        Ok(())
231    }
232
233    log_syscall!(
234        sched_setscheduler,
235        /* rv */ i32,
236        /* pid */ kernel_pid_t,
237        /* policy */ std::ffi::c_int,
238        /* param */ *const linux_api::sched::sched_attr,
239    );
240    pub fn sched_setscheduler(
241        ctx: &mut SyscallContext,
242        tid: kernel_pid_t,
243        policy: std::ffi::c_int,
244        param_ptr: ForeignPtr<sched_attr>,
245    ) -> Result<(), Errno> {
246        warn_once_then_debug!(
247            "sched_setscheduler() only updates tracked scheduler state; Shadow does not emulate Linux scheduling behavior"
248        );
249
250        let new_priority = ctx
251            .objs
252            .process
253            .memory_borrow()
254            .read(param_ptr.cast::<std::ffi::c_int>())?;
255        let reset_on_fork = (policy & SCHED_RESET_ON_FORK) != 0;
256        let policy = Sched::try_from(policy & !SCHED_RESET_ON_FORK).or(Err(Errno::EINVAL))?;
257        let policy = match policy {
258            Sched::SCHED_NORMAL
259            | Sched::SCHED_FIFO
260            | Sched::SCHED_RR
261            | Sched::SCHED_BATCH
262            | Sched::SCHED_IDLE => policy,
263            Sched::SCHED_DEADLINE => {
264                warn_once_then_debug!(
265                    "sched_setscheduler() rejects SCHED_DEADLINE because Shadow does not implement deadline scheduling semantics"
266                );
267                return Err(Errno::EINVAL);
268            }
269            Sched::SCHED_EXT => {
270                warn_once_then_debug!(
271                    "sched_setscheduler() rejects SCHED_EXT because Shadow does not implement BPF-defined scheduler semantics"
272                );
273                return Err(Errno::EINVAL);
274            }
275        };
276        validate_sched_attrs(policy, new_priority)?;
277
278        with_sched_target_thread(ctx, tid, |thread| {
279            thread.set_sched_attrs(policy, reset_on_fork, new_priority);
280        })?;
281
282        Ok(())
283    }
284
285    log_syscall!(
286        rseq,
287        /* rv */ i32,
288        /* rseq */ *const std::ffi::c_void,
289        /* rseq_len */ u32,
290        /* flags */ i32,
291        /* sig */ u32,
292    );
293    pub fn rseq(
294        ctx: &mut SyscallContext,
295        rseq_ptr: ForeignPtr<MaybeUninit<u8>>,
296        rseq_len: u32,
297        flags: std::ffi::c_int,
298        _sig: u32,
299    ) -> Result<(), Errno> {
300        // we won't need more bytes than the size of the `rseq` struct
301        let rseq_len = rseq_len.try_into().unwrap();
302        let rseq_len = std::cmp::min(rseq_len, std::mem::size_of::<rseq>());
303
304        let flags = rseq_flags::from_bits_retain(flags);
305        let unknown_flags = flags.unknown_bits();
306
307        if unknown_flags != 0 {
308            warn!("Unrecognized rseq flags: 0x{unknown_flags:x} in {flags:?}");
309            return Err(Errno::EINVAL);
310        }
311        if flags.contains(rseq_flags::RSEQ_FLAG_UNREGISTER) {
312            // TODO:
313            // * Validate that an rseq was previously registered
314            // * Validate that `sig` matches registration
315            // * Set the cpu_id of the previously registerd rseq to the uninitialized
316            //   state.
317            return Ok(());
318        }
319
320        // The `rseq` struct is designed to grow as linux needs to add more features, so we can't
321        // assume that the application making the rseq syscall is using the exact same struct as we
322        // have available in the linux_api crate (the calling application's rseq struct may have
323        // more or fewer fields). Furthermore, the rseq struct ends with a "flexible array member",
324        // which means that the rseq struct cannot be `Copy` and therefore not `Pod`.
325        //
326        // Instead, we should treat the rseq struct as a bunch of bytes and write to individual
327        // fields if possible without making assumptions about the size of the data.
328        let rseq_byte_array_ptr = ForeignArrayPtr::new(rseq_ptr, rseq_len);
329        let mut rseq_bytes = ctx
330            .objs
331            .process
332            .memory_borrow()
333            .read_vec(ForeignArrayPtr::new(rseq_ptr, rseq_len))?;
334
335        // rseq is mostly unimplemented, but also mostly unneeded in Shadow.
336        // We'd only need to implement the "real" functionality if we ever implement
337        // true preemption, in which case we'd need to do something if we ever pre-empted
338        // while the user code was in a restartable sequence. As it is, Shadow only
339        // reschedules threads at system calls, and system calls are disallowed inside
340        // restartable sequences.
341        //
342        // TODO: One place where Shadow might need to implement rseq recovery is
343        // if a hardware-based signal is delivered in the middle of an
344        // interruptible sequence.  e.g. the code in the rseq accesses an
345        // invalid address, raising SIGSEGV, but then catching it and recovering
346        // in a handler.
347        // https://github.com/shadow/shadow/issues/2139
348        //
349        // For now we just update to reflect that the thread is running on CPU 0.
350
351        let Some((cpu_id, cpu_id_start)) =
352            field_project!(&mut rseq_bytes, rseq, (cpu_id, cpu_id_start))
353        else {
354            return Err(Errno::EINVAL);
355        };
356
357        cpu_id.write(CURRENT_CPU);
358        cpu_id_start.write(CURRENT_CPU);
359
360        ctx.objs
361            .process
362            .memory_borrow_mut()
363            .copy_to_ptr(rseq_byte_array_ptr, &rseq_bytes)?;
364
365        Ok(())
366    }
367}