The Pedigree Project 0.1
PerProcessorScheduler.cc
1/*
2 * Copyright (c) 2008-2014, Pedigree Developers
3 *
4 * Please see the CONTRIB file in the root of the source tree for a full
5 * list of contributors.
6 *
7 * Permission to use, copy, modify, and distribute this software for any
8 * purpose with or without fee is hereby granted, provided that the above
9 * copyright notice and this permission notice appear in all copies.
10 *
11 * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
12 * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
13 * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
14 * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
15 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
16 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
17 * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
18 */
19
20#include "pedigree/kernel/ActivityDiagnostics.h"
21#include "pedigree/kernel/Atomic.h"
22#include "pedigree/kernel/LockGuard.h"
23#include "pedigree/kernel/Log.h"
24#include "pedigree/kernel/Metrics.h"
25#include "pedigree/kernel/Spinlock.h"
26#include "pedigree/kernel/Subsystem.h"
27#include "pedigree/kernel/debugger/commands/LocksCommand.h"
28#include "pedigree/kernel/machine/Machine.h"
29#include "pedigree/kernel/machine/SchedulerTimer.h"
30#include "pedigree/kernel/machine/Timer.h"
31#include "pedigree/kernel/machine/Trace.h"
32#include "pedigree/kernel/panic.h"
33#include "pedigree/kernel/process/Event.h"
34#include "pedigree/kernel/process/PerProcessorScheduler.h"
35#include "pedigree/kernel/process/Process.h"
36#include "pedigree/kernel/process/RoundRobin.h"
37#include "pedigree/kernel/process/Scheduler.h"
38#include "pedigree/kernel/process/SchedulingAlgorithm.h"
39#include "pedigree/kernel/process/TerminationDeferral.h"
40#include "pedigree/kernel/process/Thread.h"
41#include "pedigree/kernel/process/eventNumbers.h"
42#include "pedigree/kernel/processor/PhysicalMemoryManager.h"
43#include "pedigree/kernel/processor/Processor.h"
44#include "pedigree/kernel/processor/ProcessorInformation.h"
45#include "pedigree/kernel/processor/VirtualAddressSpace.h"
46#include "pedigree/kernel/processor/state.h"
47#include "pedigree/kernel/time/Time.h"
48#include "pedigree/kernel/utilities/utility.h"
49#if HOSTED
50#include "pedigree/kernel/processor/hosted/Processor.h"
51#endif
52#if X86_COMMON && MULTIPROCESSOR
53#include <machine/mach_pc/LocalApic.h>
54#include <machine/mach_pc/Pc.h>
55#endif
56#if PEDIGREE_HOSTED_FUNCTION_PROFILE
57#include "pedigree/kernel/processor/hosted/FunctionProfile.h"
58#endif
59
60static constexpr bool VerboseScheduler = false;
61
62namespace {
63Atomic<size_t> g_StackDiscardCount(0);
64Atomic<size_t> g_EmergencyProcessKillDiscardCount(0);
65Atomic<size_t> g_HostedRegressionDiscardCount(0);
66Atomic<size_t> g_LegacyAbiDiscardCount(0);
67} // namespace
68
70 : m_pSchedulingAlgorithm(0),
71 m_NewThreadDataLock(),
72 m_NewThreadDataCondition(),
73 m_NewThreadData(),
74 m_DelayedNewThreadData(),
75 m_NewThreadAdmissionOpen(false),
76 m_StopNewThreadWorker(false),
77 m_NewThreadWorker(),
78 m_TimeAccountingState(),
79 m_DeferredThreadReapStub(),
80 m_DeferredThreadReaps(m_DeferredThreadReapStub),
81 m_nDeferredThreadReaps(0),
82 m_DeferredThreadReapPublicationState(DeferredReapPublicationClosed),
83 m_StopTimeAccountingWorker(0),
84 m_TimeAccountingWorker(),
85 m_TimeAccountingWorkerWaiters(),
86 m_TimeAccountingWorkerWake(),
87 m_IrqWorkDoorbell(0),
88 m_ReschedulePending(0),
89 m_RemotePromptPending(0),
90 m_ClockDeadline(0),
91 m_IrqWorkLock(),
92#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
93 m_nDeferredThreadReapCompletions(0),
94#endif
95 m_pIdleThread(0) {
96}
97
98PerProcessorScheduler::~PerProcessorScheduler() {
99 // The add worker can retire a never-started detached Thread while draining.
100 // Keep ordinary destruction admission open until that producer is joined.
101 stopNewThreadWorker();
102
103 SchedulerTimer* pTimer = Machine::instance().getSchedulerTimer();
104 if (!pTimer) {
105 panic("No scheduler timer present.");
106 }
107 if (!pTimer->removeHandler(this)) {
108 FATAL("Per-processor scheduler lost timer-handler ownership.");
109 }
110
111 // With timer-driven scheduling quiesced, close publication, drain accepted
112 // targets, and finally join the ordinary destruction worker.
113 stopTimeAccountingWorker();
114}
115
116void PerProcessorScheduler::startTimeAccountingWorker(Process* pParent) {
117 if (m_TimeAccountingWorker) {
118 FATAL("Per-processor time accounting worker started twice.");
119 }
120
121 m_StopTimeAccountingWorker = 0;
122 Thread* worker = new Thread(pParent, timeAccountingWorkerEntry, this, nullptr, false, true, true);
123 worker->setName("deferred process time accounting");
124 registerWorkerWake(m_TimeAccountingWorkerWake, m_TimeAccountingWorkerWaiters);
125 m_TimeAccountingWorker.adopt(worker);
126 if (!worker->start()) {
127 FATAL("Time accounting worker could not be started.");
128 }
129
130 while (!m_DeferredThreadReapPublicationState.compareAndSwap(DeferredReapPublicationClosed, 0)) {
132 }
133 {
134 LockGuard<Spinlock> guard(m_AffinityQueueLock);
135 m_AffinityAdmissionOpen = true;
136 }
137}
138
139void PerProcessorScheduler::stopTimeAccountingWorker() {
140 if (!m_TimeAccountingWorker) {
141 return;
142 }
143 {
144 LockGuard<Spinlock> guard(m_AffinityQueueLock);
145 m_AffinityAdmissionOpen = false;
146 }
147
148 // Close allocation-free producers before allowing the worker to exit. The
149 // low bits cover a publisher which observed open admission just before this
150 // transition.
151 m_DeferredThreadReapPublicationState |= DeferredReapPublicationClosed;
152 while (m_DeferredThreadReapPublicationState.value() & DeferredReapPublicationCountMask) {
154 }
155
156 m_StopTimeAccountingWorker = 1;
157 ringIrqWorkDoorbell(m_TimeAccountingWorkerWake);
159 m_TimeAccountingWorker.join();
160 unregisterWorkerWake(m_TimeAccountingWorkerWake);
161 if (m_nDeferredThreadReaps.value() || m_AffinityRequests.value()) {
162 FATAL("Deferred Thread reap worker stopped with pending targets.");
163 }
164}
165
166int PerProcessorScheduler::timeAccountingWorkerEntry(void* instance) {
167 return reinterpret_cast<PerProcessorScheduler*>(instance)->runTimeAccountingWorker();
168}
169
170int PerProcessorScheduler::runTimeAccountingWorker() {
171 TerminationDeferral workerLifetime;
172 while (true) {
173 const bool stopping = m_StopTimeAccountingWorker.value() != 0;
174 const bool pending = m_TimeAccountingState.ready() || m_nDeferredThreadReaps.value() ||
175 m_AffinityRequests.value();
176 if (stopping && m_TimeAccountingState.caughtUp() && !m_nDeferredThreadReaps.value() &&
177 !m_AffinityRequests.value()) {
178 break;
179 }
180 if (!pending) {
181 auto guard = m_TimeAccountingWorkerWaiters.acquire();
182 const bool stillPending = m_TimeAccountingState.ready() || m_nDeferredThreadReaps.value() ||
183 m_AffinityRequests.value() || m_StopTimeAccountingWorker.value();
184 if (!stillPending) {
185 const WaitQueue::WakeReason reason =
186 guard.wait(WaitQueue::Channel(), Thread::CondWait, reinterpret_cast<uintptr_t>(this));
187 (void)reason;
188 }
189 continue;
190 }
191
192 const size_t target = m_TimeAccountingState.beginBatch();
195 drainDeferredThreadReaps();
196 drainAffinityRequests();
197 // Sampling can be preempted while owning its global mutex. Keep this
198 // worker accounted until every operation in the batch has retired.
199 m_TimeAccountingState.finishBatch(target);
200
201 // Give ordinary peers a scheduling turn before draining another batch.
203 }
204
205 return 0;
206}
207
208void PerProcessorScheduler::publishDeferredThreadReap(Thread* thread) {
209 const size_t admission = (m_DeferredThreadReapPublicationState += 1);
210 if (admission & DeferredReapPublicationClosed) {
211 m_DeferredThreadReapPublicationState -= 1;
212 FATAL_NOLOCK("Deferred Thread reap publication reached a stopped worker.");
213 }
214
216 if (node.thread != thread) {
217 FATAL_NOLOCK("Deferred Thread reap node has invalid ownership.");
218 }
219
220 // Make the worker wake edge visible before the node is consumable. A
221 // transient pop simply leaves the nonzero count visible for its next turn.
222 m_nDeferredThreadReaps += 1;
223 m_DeferredThreadReaps.push(node);
224 ringIrqWorkDoorbell(m_TimeAccountingWorkerWake);
225 m_DeferredThreadReapPublicationState -= 1;
226}
227
228bool PerProcessorScheduler::drainDeferredThreadReaps() {
229 using PopResult =
230 IntrusiveMpscQueue<DeferredThreadReapNode, &DeferredThreadReapNode::next>::PopResult;
231
232 while (true) {
233 DeferredThreadReapNode* node = nullptr;
234 const PopResult result = m_DeferredThreadReaps.pop(node);
235 if (result == PopResult::Empty) {
236 return true;
237 }
238 if (result == PopResult::Transient) {
239 return false;
240 }
241 if (!node || !node->thread) {
242 FATAL("Deferred Thread reap queue returned an invalid target.");
243 }
244
245 Thread* thread = node->thread;
246 Process* process = thread->getParent();
247 if (Processor::executionContext() != ExecutionContext::WaitableThread ||
249 FATAL("Deferred Thread reap worker is not at an IRQ-enabled WaitableThread boundary.");
250 }
251
252 delete thread;
253 process->m_DeferredThreadReaps.leave();
254 m_nDeferredThreadReaps -= 1;
255#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
256 m_nDeferredThreadReapCompletions += 1;
257#endif
258 }
259}
260
262 Thread* pThread;
263 Thread::ThreadStartFunc pStartFunction;
264 void* pParam;
265 bool bUsermode;
266 void* pStack;
267 SyscallState state;
268 bool useSyscallState;
269};
270
271void PerProcessorScheduler::startNewThreadWorker(Process* pParent) {
272 m_NewThreadDataLock.acquire();
273 if (m_NewThreadWorker || m_NewThreadData.count() || m_DelayedNewThreadData.count()) {
274 m_NewThreadDataLock.release();
275 FATAL("Per-processor thread add worker started with live state.");
276 }
277 m_StopNewThreadWorker = false;
278 m_NewThreadAdmissionOpen = true;
279 m_NewThreadDataLock.release();
280
281 Thread* pAddThread =
282 new Thread(pParent, processorAddThread, reinterpret_cast<void*>(this), 0, false, true);
283 pAddThread->setName("PerProcessorScheduler thread add worker");
284 m_NewThreadWorker.adopt(pAddThread);
285}
286
287void PerProcessorScheduler::stopNewThreadWorker() {
288 m_NewThreadDataLock.acquire();
289 m_NewThreadAdmissionOpen = false;
290 m_StopNewThreadWorker = true;
291 while (m_DelayedNewThreadData.count()) {
292 m_NewThreadData.pushBack(m_DelayedNewThreadData.popFront());
293 }
294 const bool pending = m_NewThreadData.count();
295 m_NewThreadDataLock.release();
296
297 if (!m_NewThreadWorker) {
298 if (pending) {
299 FATAL("Per-processor thread add queue has no worker.");
300 }
301 return;
302 }
303
304 m_NewThreadDataCondition.broadcast();
305 m_NewThreadWorker.join();
306
307 m_NewThreadDataLock.acquire();
308 const bool drained = !m_NewThreadData.count() && !m_DelayedNewThreadData.count();
309 m_NewThreadDataLock.release();
310 if (!drained) {
311 FATAL("Per-processor thread add worker stopped before draining.");
312 }
313}
314
315int PerProcessorScheduler::processorAddThread(void* instance) {
316 PerProcessorScheduler* pInstance = reinterpret_cast<PerProcessorScheduler*>(instance);
317 while (true) {
318 pInstance->m_NewThreadDataLock.acquire();
319 while (!pInstance->m_NewThreadData.count()) {
320 if (pInstance->m_StopNewThreadWorker) {
321 pInstance->m_NewThreadDataLock.release();
322 return 0;
323 }
324
325 pInstance->m_NewThreadDataCondition.waitForCompletion(pInstance->m_NewThreadDataLock);
326 }
327
328 void* p = pInstance->m_NewThreadData.popFront();
329
330 newThreadData* pData = reinterpret_cast<newThreadData*>(p);
331
332 if (pInstance != &Processor::information().getScheduler()) {
333 pInstance->m_NewThreadDataLock.release();
334 FATAL("instance " << instance << " does not match current scheduler in processorAddThread!");
335 }
336
337 Thread* pThread = pData->pThread;
338 pThread->m_Lock.acquire();
339 const bool retireBeforeStart = pThread->getUnwindState() == Thread::TerminateThread;
340 if (pThread->m_Status == Thread::Created && pThread->m_bStartRequested && !retireBeforeStart) {
341 pThread->m_bStartRequested = false;
342 pThread->m_Status = Thread::Ready;
343 }
344
345 const bool runnable =
346 pThread->m_Status == Thread::Running || pThread->m_Status == Thread::Ready;
347 if (retireBeforeStart) {
348 pThread->m_Lock.release();
349 pInstance->m_NewThreadDataLock.release();
350 // This thread has never owned a running stack. The add worker owns
351 // the last queued reference and can complete its off-stack exit.
352 delete pData;
353 pThread->shutdown();
354 pThread->m_Lock.acquire();
355 EMIT_IF(TRACK_LOCKS) {
356 g_LocksCommand.lockReleased(&pThread->m_Lock);
357 }
358 deleteThread(pThread);
359 continue;
360 }
361
362 if (!runnable) {
363 if (pThread->m_Status != Thread::Created) {
364 pThread->m_Lock.release();
365 pInstance->m_NewThreadDataLock.release();
366 FATAL(
367 "Per-processor add worker cannot park an already "
368 "scheduled thread.");
369 }
370
371 // Keep the queue and thread state serialized until the parked record
372 // is visible, so start or termination cannot miss its publication.
373 const bool stopping = pInstance->m_StopNewThreadWorker;
374 if (!stopping) {
375 pInstance->m_DelayedNewThreadData.pushBack(p);
376 }
377 pThread->m_Lock.release();
378 pInstance->m_NewThreadDataLock.release();
379 if (!stopping) {
380 continue;
381 }
382
384 delete pData;
385 pThread->shutdown();
386 pThread->m_Lock.acquire();
387 EMIT_IF(TRACK_LOCKS) {
388 g_LocksCommand.lockReleased(&pThread->m_Lock);
389 }
390 deleteThread(pThread);
391 continue;
392 }
393
394 pInstance->m_NewThreadDataLock.release();
395 pThread->setCpuId(Processor::id());
396 if (pData->useSyscallState) {
397 pInstance->addThread(pThread, pData->state);
398 } else {
399 pInstance->addThread(pThread, pData->pStartFunction, pData->pParam, pData->bUsermode,
400 pData->pStack);
401 }
402 delete pData;
403 }
404}
405
407 m_LogicalCpu = Processor::index();
408 m_PhysicalCpu = Processor::id();
409 // Bootstrap identity accessors still return zero here. The discovered
410 // topology already distinguishes its logical slot from the firmware ID.
411 for (size_t cpu = 0; cpu < Processor::getCount(); ++cpu) {
412 ProcessorInformation* information = Processor::informationAt(cpu);
413 if (information && &information->getScheduler() == this) {
414 m_LogicalCpu = cpu;
415 m_PhysicalCpu = information->processorId();
416 break;
417 }
418 }
420
421 if (!pThread->m_Placement.migratable) {
422 pThread->m_Placement.allowed = CpuAffinityMask();
423 pThread->m_Placement.allowed.set(m_LogicalCpu);
424 }
425 pThread->m_HasSchedulerContext = true;
426 pThread->setStatus(Thread::Running);
427 pThread->setCpuId(m_PhysicalCpu);
428 Processor::information().setCurrentThread(pThread);
429 pThread->recordTime(CpuTimeMode::Kernel);
430
432 Processor::information().setKernelStack(reinterpret_cast<uintptr_t>(pThread->getKernelStack()));
434
435 startNewThreadWorker(pThread->getParent());
436 startTimeAccountingWorker(pThread->getParent());
437
438 // Do not publish timer-driven exit producers until their allocation-free
439 // retirement consumer is live and accepting work.
440 SchedulerTimer* pTimer = Machine::instance().getSchedulerTimer();
441 if (!pTimer) {
442 panic("No scheduler timer present.");
443 }
444 if (!pTimer->registerHandler(this)) {
445 FATAL("Per-processor scheduler timer handler is already owned.");
446 }
447 m_NominalQuantumNs = pTimer->nominalQuantumNs();
448 m_OneShotTimer = pTimer->supportsOneShot();
449 if (m_OneShotTimer) {
450 if (m_LogicalCpu == 0) {
451 m_NextLoadSampleDeadline = Time::getTicksFast() + LoadAverage::PeriodNanoseconds;
452 }
453 updateOneShotTimer();
454 }
455}
456
457void PerProcessorScheduler::programOneShotTimer() {
458 if (!m_OneShotTimer) {
459 return;
460 }
461 uint64_t deadline = m_QuantumDeadline;
462 if (m_NextLoadSampleDeadline && (!deadline || m_NextLoadSampleDeadline < deadline)) {
463 deadline = m_NextLoadSampleDeadline;
464 }
465 const uint64_t clockDeadline = m_ClockDeadline.value();
466 if (clockDeadline && (!deadline || clockDeadline < deadline)) {
467 deadline = clockDeadline;
468 }
469 SchedulerTimer* timer = Machine::instance().getSchedulerTimer();
470 if (deadline) {
471 if (!timer->armDeadline(deadline)) {
472 FATAL_NOLOCK("Failed to arm the local scheduler deadline.");
473 }
474 } else {
475 timer->disarm();
476 }
477}
478
480 if (!m_OneShotTimer) {
481 return;
482 }
483 uint64_t previous = m_ClockDeadline.value();
484 while (!m_ClockDeadline.compareAndSwap(previous, deadline)) {
485 previous = m_ClockDeadline.value();
486 }
487 if ((!previous && !deadline) || (previous && (!deadline || deadline >= previous))) {
488 // An already programmed earlier interrupt will re-evaluate the deadline.
489 return;
490 }
491 if (this == &Processor::information().getScheduler()) {
492 const bool interrupts = Processor::getInterrupts();
494 programOneShotTimer();
495 Processor::setInterrupts(interrupts);
496 return;
497 }
498#if X86_COMMON && MULTIPROCESSOR
499 ProcessorInformation* information = Processor::informationAt(m_LogicalCpu);
500 assert(information);
501 if (!Pc::instance().getLocalApic().interProcessorInterrupt(
502 information->localApicId(), IPI_RESCHEDULE_VECTOR, LocalApic::deliveryModeFixed, true,
503 false)) {
504 FATAL_NOLOCK("Remote clock deadline rearm failed.");
505 }
506#endif
507}
508
509void PerProcessorScheduler::updateOneShotTimer() {
510 if (!m_OneShotTimer) {
511 return;
512 }
513 const bool interrupts = Processor::getInterrupts();
515 Thread* current = Processor::information().getCurrentThread();
516 if (current && current != m_pIdleThread && m_pSchedulingAlgorithm->hasReady()) {
517 const uint64_t now = Time::getTicksFast();
518 m_QuantumDeadline =
519 now > ~uint64_t(0) - m_NominalQuantumNs ? ~uint64_t(0) : now + m_NominalQuantumNs;
520 } else {
521 m_QuantumDeadline = 0;
522 }
523 programOneShotTimer();
524 Processor::setInterrupts(interrupts);
525}
526
527void PerProcessorScheduler::armLocalQuantumIfNeeded() {
528 if (!m_OneShotTimer || m_QuantumDeadline || this != &Processor::information().getScheduler()) {
529 return;
530 }
531 const bool interrupts = Processor::getInterrupts();
533 Thread* current = Processor::information().getCurrentThread();
534 if (current && current != m_pIdleThread && !m_QuantumDeadline) {
535 const uint64_t now = Time::getTicksFast();
536 m_QuantumDeadline =
537 now > ~uint64_t(0) - m_NominalQuantumNs ? ~uint64_t(0) : now + m_NominalQuantumNs;
538 programOneShotTimer();
539 }
540 Processor::setInterrupts(interrupts);
541}
542
543void PerProcessorScheduler::schedule(Thread::Status nextStatus, bool dispatchEvents) {
544 if (!Processor::guardDeviceHardIrqOperation(DeviceHardIrqOperation::Schedule)) {
545 return;
546 }
547
548 bool bWasInterrupts = Processor::getInterrupts();
550
551 // The explicit affinity gate may have moved after its caller selected a
552 // scheduler. Resolve the receiver only once IRQs exclude another handoff.
553 Processor::information().getScheduler().scheduleWithInterruptState(nextStatus, dispatchEvents,
554 bWasInterrupts);
555}
556
557void PerProcessorScheduler::scheduleWithInterruptState(Thread::Status nextStatus,
558 bool dispatchEvents, bool bWasInterrupts) {
559#if PEDIGREE_HOSTED_FUNCTION_PROFILE
560 hostedFunctionProfileInvalidate(HostedProfileInvalidation::Schedule);
561#endif
562 assert(!Processor::getInterrupts());
563 ActivityDiagnostics::recordScheduleCall();
564 Metrics::increment(Metrics::Counter::Schedule);
565
566 Thread* pCurrentThread = Processor::information().getCurrentThread();
567 if (!pCurrentThread) {
568 FATAL("Missing a current thread in PerProcessorScheduler::schedule!");
569 }
570
571 bool canServiceWorkerWakeups = bWasInterrupts;
572#if HOSTED
573 canServiceWorkerWakeups &= !pCurrentThread->getHostedSignalDepth();
574#endif
575 if (canServiceWorkerWakeups && m_IrqWorkDoorbell.compareAndSwap(1, 0)) {
576 serviceWorkerWakeups();
577 }
578
579 // Grab the current thread's lock.
580 pCurrentThread->getLock().acquire();
581
582 bool dispatchEvent = false;
583 if (nextStatus == Thread::Sleeping) {
584 if (!pCurrentThread->hasActiveWaitUnlocked()) {
585 if (!pCurrentThread->consumeTerminalWaitCancelledBeforeBlockUnlocked()) {
586 FATAL("Scheduler refused a sleep without an active WaitQueue.");
587 }
588
589 // A terminal cancellation can unlink an already-published waiter
590 // before this thread has committed its Sleeping transition. The
591 // one-shot handoff above is the only valid unlinked sleep.
592 pCurrentThread->getLock().release();
593 Processor::setInterrupts(bWasInterrupts);
594 return;
595 }
596
597 // The wait record is published before blockCurrent(). A wake in that
598 // window changes it away from Waiting, so it is impossible to commit a
599 // stale Sleeping transition.
600 dispatchEvent = pCurrentThread->hasDeliverableEventsUnlocked();
601 if (dispatchEvent) {
602 PerProcessorScheduler* readyScheduler = nullptr;
603 pCurrentThread->interruptWaitUnlocked(WaitQueue::WakeReason::Event, readyScheduler);
604 }
605 if (!pCurrentThread->activeWaitPendingUnlocked() || dispatchEvent) {
606 pCurrentThread->getLock().release();
607 Processor::setInterrupts(bWasInterrupts);
608 return;
609 }
610 }
611
612 // Now attempt to get another thread to run.
613 // This will also get the lock for the returned thread.
614 Thread* pNextThread = selectNext(pCurrentThread);
615 if (pNextThread == 0) {
616 ActivityDiagnostics::recordSchedulerIdleFallback(
617 pCurrentThread->m_Status == Thread::Ready,
618 __atomic_load_n(&pCurrentThread->m_ReadyPublicationPending, __ATOMIC_ACQUIRE));
619 // A tick or yield does not make a runnable thread idle. Workers which have
620 // no work park themselves on their WaitQueue before reaching this path.
621 if (nextStatus == Thread::Ready && pCurrentThread != m_pIdleThread &&
622 !pCurrentThread->m_ReadyPublicationPending) {
623 pNextThread = pCurrentThread;
624 } else if (m_pIdleThread == 0) {
625 // The scheduler is still bootstrapping, so spinning is the only
626 // available fallback.
627 pCurrentThread->getLock().release();
628 Processor::setInterrupts(bWasInterrupts);
629 return;
630 } else {
631 // Another CPU may publish ready work after selectNext releases the
632 // queue lock. It remains queued for the next scheduling interrupt.
633 pNextThread = m_pIdleThread;
634 if (pNextThread != pCurrentThread)
635 pNextThread->getLock().acquire();
636 }
637 }
638
639 // Saving and restoring the same hosted context does not yield and can
640 // strand the add-thread worker, so return directly when current stays on CPU.
641 if (pNextThread == pCurrentThread) {
642 ActivityDiagnostics::recordSameThreadSelection();
643 Metrics::increment(Metrics::Counter::SameThread);
644 updateOneShotTimer();
645 const bool waitOwnsEventDispatch = pCurrentThread->hasActiveWaitUnlocked();
646 pCurrentThread->getLock().release();
647 Processor::setInterrupts(bWasInterrupts);
648 if (dispatchEvents && !waitOwnsEventDispatch) {
649 Processor::information().getScheduler().checkEventState(0);
650 }
651 return;
652 }
653
654 EMIT_IF(VerboseScheduler) {
655 NOTICE_NOLOCK("schedule: " << pCurrentThread << " -> " << pNextThread << " -- "
656 << pCurrentThread->getName() << " -> " << pNextThread->getName());
657 }
658
659 // Now neither thread can be moved, we're safe to switch.
660 ActivityDiagnostics::recordContextSwitch();
661 Metrics::increment(Metrics::Counter::ContextSwitch);
662 if (pNextThread == m_pIdleThread) {
663 ActivityDiagnostics::recordIdleSelection();
664 Metrics::increment(Metrics::Counter::IdleSelection);
665 }
666 if (pCurrentThread != m_pIdleThread)
667 pCurrentThread->setStatusUnlocked(nextStatus);
668 pNextThread->setStatusUnlocked(Thread::Running);
669 Processor::information().setCurrentThread(pNextThread);
670 updateOneShotTimer();
671
672 // Load the new kernel stack into the TSS, and the new TLS base and switch
673 // address spaces
674 Processor::information().setKernelStack(
675 reinterpret_cast<uintptr_t>(pNextThread->getKernelStack()));
677 Processor::setTlsBase(pNextThread->getTlsBase());
678
679 // Update times.
680 pCurrentThread->trackTime(CpuTimeMode::Kernel);
681 pNextThread->recordTime(CpuTimeMode::Kernel);
682
683 pNextThread->getLock().release();
684
685 // The real switch releases the old current thread's lock after changing
686 // stacks, so retire that deferred release from the lock checker now.
687 EMIT_IF(TRACK_LOCKS) {
688 g_LocksCommand.lockReleased(&pCurrentThread->getLock());
689 }
690
691 EMIT_IF(TRACK_LOCKS) {
692 if (!g_LocksCommand.checkSchedule()) {
693 FATAL("Lock checker disallowed this reschedule.");
694 }
695 }
696
697 EMIT_IF(SYSTEM_REQUIRES_ATOMIC_CONTEXT_SWITCH) {
698 Processor::switchState(bWasInterrupts, pCurrentThread->state(), pNextThread->state(),
699 pCurrentThread->getLock().deferredReleaseWord());
700 const bool waitOwnsEventDispatch = pCurrentThread->hasActiveWaitUnlocked();
701#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
702 Processor::notifyHostedContextSwitchStage(
703 ProcessorBase::HostedContextSwitchStage::SchedulerBookkeepingComplete);
704 Processor::notifyHostedContextSwitchStage(
705 ProcessorBase::HostedContextSwitchStage::SchedulerRestoringInterrupts);
706#endif
707 Processor::setInterrupts(bWasInterrupts);
708 if (dispatchEvents && !waitOwnsEventDispatch) {
709 Processor::information().getScheduler().checkEventState(0);
710 }
711 }
712 else {
713 // NOTICE_NOLOCK("calling saveState [schedule]");
714 if (Processor::saveState(pCurrentThread->state())) {
715 // Just context-restored, return.
716
717 // A resumed WaitQueue must retire its outer wait record before an
718 // event handler can enter another blocking operation. WaitQueue
719 // performs the event check immediately after that retirement.
720 const bool waitOwnsEventDispatch = pCurrentThread->hasActiveWaitUnlocked();
721
722 // Return to previous interrupt state.
723 Processor::setInterrupts(bWasInterrupts);
724 if (dispatchEvents && !waitOwnsEventDispatch) {
725 // We don't have a user-mode stack available here, so pass zero
726 // and don't execute user-mode event handlers.
727 Processor::information().getScheduler().checkEventState(0);
728 }
729
730 return;
731 }
732
733 // Restore context, releasing the old thread's lock when we've switched
734 // stacks.
735 Processor::restoreState(pNextThread->state(), pCurrentThread->getLock().deferredReleaseWord());
736 // Not reached.
737 }
738}
739
740void PerProcessorScheduler::checkEventState(uintptr_t userStack) {
741 checkEventState(userStack, Thread::EventSelection::WithoutExactUserReturn);
742}
743
744void PerProcessorScheduler::checkEventState(uintptr_t userStack, Thread::EventSelection selection) {
745 checkEventState(userStack, selection, nullptr, nullptr);
746}
747
748void PerProcessorScheduler::checkEventState(uintptr_t userStack, Thread::EventSelection selection,
749 InterruptState* interruptState,
750 SyscallState* syscallState) {
751 bool bWasInterrupts = Processor::getInterrupts();
753
754 size_t pageSz = PhysicalMemoryManager::getPageSize();
755
756 Thread* pThread = Processor::information().getCurrentThread();
757 if (!pThread) {
758 Processor::setInterrupts(bWasInterrupts);
759 return;
760 }
761
762 if (pThread->getScheduler() != this) {
763 // Wrong scheduler - don't try to run an event for this thread.
764 Processor::setInterrupts(bWasInterrupts);
765 return;
766 }
767
768 if (pThread->eventsDeferred()) {
769 // Cannot check for any events - we aren't allowed to handle them.
770 Processor::setInterrupts(bWasInterrupts);
771 return;
772 }
773
774 if (selection == Thread::EventSelection::WithoutExactUserReturn) {
776 }
777
778 Event::Delivery eventDelivery = pThread->getNextEvent(selection);
779 Event* pEvent = eventDelivery.get();
780 if (!eventDelivery) {
781 Processor::setInterrupts(bWasInterrupts);
782 return;
783 }
784
785 UserReturnFrame* frame = pThread->currentUserReturnFrame();
786 Subsystem* subsystem = pThread->getParent() ? pThread->getParent()->getSubsystem() : nullptr;
787 if (frame && subsystem && pEvent->isSignalEvent() && pEvent->getNumber() != 9) {
788 if (!bWasInterrupts)
789 FATAL_NOLOCK("User-return event interception requires IRQ-enabled entry");
790 Subsystem::UserReturnEventResult intercepted;
791 {
792 // A tracing stop may be killed without returning through this stack.
793 Thread::StackDiscardScope deliveryDiscard(
794 [](void* value) { static_cast<Event::Delivery*>(value)->reset(); }, &eventDelivery);
796 intercepted = subsystem->userReturnEvent(*pThread, *pEvent, *frame);
798 }
799 if (intercepted != Subsystem::UserReturnEventResult::Deliver) {
800 frame->m_Terminal = intercepted == Subsystem::UserReturnEventResult::Terminal;
801 eventDelivery.reset();
802 Processor::setInterrupts(bWasInterrupts);
803 return;
804 }
805 }
806
807 if (pEvent->requiresExactUserReturnState()) {
808 Event::UserReturnDelivery result = Event::UserReturnDelivery::NotApplicable;
809 if (interruptState || syscallState) {
810 if (!bWasInterrupts) {
811 FATAL_NOLOCK("Exact user-return event delivery requires an IRQ-enabled thread boundary.");
812 }
813
814 // Signal-frame construction uses the guarded user-copy path, which may
815 // wait for a VM operation already in flight. The Delivery lease pins the
816 // event while the raw return frame stays on this thread's kernel stack.
818 if (interruptState) {
819 result = pEvent->deliverAtUserReturn(*interruptState);
820 } else {
821 result = pEvent->deliverAtUserReturn(*syscallState);
822 }
824 }
825
826 if (result == Event::UserReturnDelivery::NotApplicable) {
827 // Keep the delivery pending until an architecture return tail supplies
828 // the exact user register image it needs.
829 pThread->sendEvent(pEvent);
830 }
831 eventDelivery.reset();
832 Processor::setInterrupts(bWasInterrupts);
833 return;
834 }
835
836 uintptr_t handlerAddress = pEvent->getHandlerAddress();
837
838 VirtualAddressSpace& va = Processor::information().getVirtualAddressSpace();
839 physical_uintptr_t page = 0;
840 size_t flags = 0;
841 bool mappingAvailable = true;
842#if HOSTED
843 if (pEvent->getHandlerPrivilege() == Event::HandlerPrivilege::Kernel) {
845 } else {
846 mappingAvailable = va.isMapped(reinterpret_cast<void*>(handlerAddress));
847 if (mappingAvailable) {
848 va.getMapping(reinterpret_cast<void*>(handlerAddress), page, flags);
849 }
850 }
851#else
852 mappingAvailable = va.isMapped(reinterpret_cast<void*>(handlerAddress));
853 if (mappingAvailable) {
854 va.getMapping(reinterpret_cast<void*>(handlerAddress), page, flags);
855 }
856#endif
857
858 const bool userHandler = pEvent->getHandlerPrivilege() == Event::HandlerPrivilege::User;
859 const bool userAddress =
860 !userHandler ||
861 (handlerAddress >= va.getUserStart() && handlerAddress < va.getKernelStart() &&
862 va.isAddressValid(reinterpret_cast<void*>(handlerAddress)));
863 if (!mappingAvailable || !userAddress || !pEvent->isValidHandlerMapping(flags)) {
864 ERROR_NOLOCK("checkEventState: Handler address "
865 << Hex << handlerAddress << " does not match its declared "
866 << (userHandler ? "user" : "kernel") << " privilege.");
867 Processor::setInterrupts(bWasInterrupts);
868 return;
869 }
870
871 bool alternateStackCandidate = false;
872 uintptr_t alternateStackTop = 0;
873 if (userHandler && pEvent->prefersAlternateUserStack()) {
875 if (alternate.enabled && !alternate.inUse && alternate.base &&
876 alternate.size <= (~static_cast<uintptr_t>(0) - alternate.base)) {
877 alternateStackTop = (alternate.base + alternate.size) & ~static_cast<uintptr_t>(0xF);
878 alternateStackCandidate = true;
879 }
880 }
881
882 SchedulerState* oldState = pThread->pushState();
883 if (!oldState) {
884 // Keep the event pending until an outer handler unwinds and makes a
885 // state slot available.
886 pThread->sendEvent(pEvent);
887 Processor::setInterrupts(bWasInterrupts);
888 return;
889 }
890
891 if (userHandler) {
892 if (alternateStackCandidate) {
893 userStack = alternateStackTop;
894 }
895
896 bool usableUserStack = false;
897 if (userStack >= pageSz) {
898 const uintptr_t stackPage = userStack - pageSz;
899 if (stackPage >= va.getUserStart() && stackPage < va.getKernelStart() &&
900 va.isAddressValid(reinterpret_cast<void*>(stackPage)) &&
901 va.isMapped(reinterpret_cast<void*>(stackPage))) {
902 va.getMapping(reinterpret_cast<void*>(stackPage), page, flags);
903 usableUserStack = !(flags & VirtualAddressSpace::KernelMode);
904 }
905 }
906
907 if (usableUserStack && alternateStackCandidate) {
908 pThread->m_AlternateSignalStack.inUse = true;
909 pThread->m_StateLevels[pThread->m_nStateLevel].m_bOwnsAlternateSignalStack = true;
910 }
911
912 if (!usableUserStack) {
913 VirtualAddressSpace::Stack* stateStack = pThread->getStateUserStack();
914 if (!stateStack || !va.isMapped(adjust_pointer(stateStack->getTop(), -pageSz))) {
915 stateStack = va.allocateStack();
916 if (!stateStack)
917 panic("checkEventState: no user fallback stack");
918 pThread->setStateUserStack(stateStack);
919 }
920
921 userStack = reinterpret_cast<uintptr_t>(stateStack->getTop());
922 }
923 }
924
925 // The address of the serialize buffer is determined by the thread ID and
926 // the nesting level.
927 uintptr_t addr = Event::getHandlerBuffer() +
928 (pThread->getId() * MAX_NESTED_EVENTS + (pThread->getStateLevel() - 1)) *
930
931 // Ensure the complete serialized-event span is mapped.
932 for (size_t offset = 0; offset < Event::getHandlerBufferSize(); offset += pageSz) {
933 void* eventPage = reinterpret_cast<void*>(addr + offset);
934 if (!va.isMapped(eventPage)) {
935 physical_uintptr_t p = PhysicalMemoryManager::instance().allocatePage();
936 if (!p) {
937 panic("checkEventState: Out of memory!");
938 }
940 }
941 }
942
943 const bool deletableEvent = pEvent->isDeletable();
944 if (!deletableEvent && !userHandler) {
945 eventDelivery.beginDispatch();
946 }
947 pEvent->serialize(reinterpret_cast<uint8_t*>(addr));
948
949 // Fire-and-forget events are fully represented by their serialized data.
950 // Stable kernel events retain the lease through their callback because
951 // that callback may intentionally refer back to the original object.
952 if (deletableEvent) {
953 eventDelivery.reset();
954 }
955
956 EMIT_IF(!SYSTEM_REQUIRES_ATOMIC_CONTEXT_SWITCH) {
957 if (Processor::saveState(*oldState)) {
958 // Just context-restored.
959 Processor::setInterrupts(bWasInterrupts);
960 return;
961 }
962 }
963
964 if (!userHandler) {
965 // Setup must be atomic, but the callback is ordinary thread work. In
966 // particular, a return-to-user interrupt tail enters here only after
967 // raw handler/accounting scopes have unwound and with IRQs enabled.
968 Processor::setInterrupts(bWasInterrupts);
969#if HOSTED
970 // Hosted state levels use distinct signal/scheduler stacks. Run the
971 // handler on the selected level so an interrupt cannot save a
972 // level-one frame that is physically still on level zero.
973 callOnStack(reinterpret_cast<uintptr_t>(pThread->getKernelStack()), handlerAddress, addr);
974#else
975 void (*fn)(size_t) = reinterpret_cast<void (*)(size_t)>(handlerAddress);
976 fn(addr);
977#endif
978
980 eventDelivery.reset();
981 pThread->popState(!HOSTED);
982 Processor::setInterrupts(bWasInterrupts);
983 return;
984 } else if (userStack != 0) {
985 // User delivery consumes only the serialized representation.
986 eventDelivery.reset();
987 pThread->transitionTime(CpuTimeMode::Kernel, CpuTimeMode::User);
988 EMIT_IF(SYSTEM_REQUIRES_ATOMIC_CONTEXT_SWITCH) {
989 Processor::saveAndJumpUser(bWasInterrupts, *oldState, 0, Event::getTrampoline(), userStack,
990 handlerAddress, addr);
991 Processor::setInterrupts(bWasInterrupts);
992 }
993 else {
994 Processor::jumpUser(0, Event::getTrampoline(), userStack, handlerAddress, addr);
995 // Not reached.
996 }
997 }
998}
999
1002
1003 Thread* pThread = Processor::information().getCurrentThread();
1004 pThread->abandonCurrentState(false);
1005
1006 Processor::restoreState(pThread->state());
1007 // Not reached.
1008}
1009
1011 void* pParam, bool bUsermode, void* pStack) {
1012 // Handle wrong CPU, and handle thread not yet ready to schedule.
1013 if (this != &Processor::information().getScheduler() || pThread->getStatus() == Thread::Created) {
1014 newThreadData* pData = new newThreadData;
1015 pData->pThread = pThread;
1016 pData->pStartFunction = pStartFunction;
1017 pData->pParam = pParam;
1018 pData->bUsermode = bUsermode;
1019 pData->pStack = pStack;
1020 pData->useSyscallState = false;
1021
1022 pThread->m_Lock.release();
1023
1024 m_NewThreadDataLock.acquire();
1025 if (!m_NewThreadAdmissionOpen) {
1026 m_NewThreadDataLock.release();
1027 delete pData;
1028 panic("Thread admitted after its per-processor worker stopped.");
1029 }
1030 m_NewThreadData.pushBack(pData);
1031 m_NewThreadDataLock.release();
1032
1033 m_NewThreadDataCondition.signal();
1034
1035 return;
1036 }
1037
1038 pThread->setCpuId(Processor::id());
1039 pThread->setScheduler(this);
1040
1041 bool bWasInterrupts = Processor::getInterrupts();
1043
1044 // We assume here that pThread's lock is already taken.
1045
1046 Thread* pCurrentThread = Processor::information().getCurrentThread();
1047
1048 // Grab the current thread's lock.
1049 pCurrentThread->getLock().acquire();
1050
1052
1053 assert(pThread->m_Placement.allowed.contains(m_LogicalCpu));
1054 pThread->m_HasSchedulerContext = true;
1055 // Now neither thread can be moved, we're safe to switch.
1056 Metrics::increment(Metrics::Counter::ContextSwitch);
1057 if (pThread == m_pIdleThread) {
1058 Metrics::increment(Metrics::Counter::IdleSelection);
1059 }
1060 if (pCurrentThread != m_pIdleThread) {
1061 pCurrentThread->setStatusUnlocked(Thread::Ready);
1062 }
1063 pThread->setStatusUnlocked(Thread::Running);
1064 Processor::information().setCurrentThread(pThread);
1065 updateOneShotTimer();
1066 void* kernelStack = pThread->getKernelStack();
1067 Processor::information().setKernelStack(reinterpret_cast<uintptr_t>(kernelStack));
1070
1071 // This thread is safe from being moved as its status is now "running".
1072 // It is worth noting that we can't just call exit() here, as the lock is
1073 // not necessarily actually taken.
1074 if (pThread->getLock().interrupts())
1075 bWasInterrupts = true;
1076 bool bWas = pThread->getLock().acquired();
1077 pThread->getLock().unlockForScheduler();
1078 EMIT_IF(TRACK_LOCKS) {
1079 // Satisfy the lock checker; we're releasing these out of order, so make
1080 // sure the checker sees them unlocked in order.
1081 g_LocksCommand.lockReleased(&pCurrentThread->getLock());
1082 if (bWas) {
1083 // Lock was in fact locked before.
1084 g_LocksCommand.lockReleased(&pThread->getLock());
1085 }
1086 if (!g_LocksCommand.checkSchedule()) {
1087 FATAL("Lock checker disallowed this reschedule.");
1088 }
1089 }
1090
1091 pCurrentThread->trackTime(CpuTimeMode::Kernel);
1092 pThread->recordTime(bUsermode ? CpuTimeMode::User : CpuTimeMode::Kernel);
1093
1094 EMIT_IF(SYSTEM_REQUIRES_ATOMIC_CONTEXT_SWITCH) {
1095 if (bUsermode) {
1097 bWasInterrupts, pCurrentThread->state(), pCurrentThread->getLock().deferredReleaseWord(),
1098 reinterpret_cast<uintptr_t>(pStartFunction), reinterpret_cast<uintptr_t>(pStack),
1099 reinterpret_cast<uintptr_t>(pParam));
1100 } else {
1102 bWasInterrupts, pCurrentThread->state(), pCurrentThread->getLock().deferredReleaseWord(),
1103 reinterpret_cast<uintptr_t>(pStartFunction), reinterpret_cast<uintptr_t>(pStack),
1104 reinterpret_cast<uintptr_t>(pParam));
1105 }
1106 Processor::setInterrupts(bWasInterrupts);
1107 }
1108 else {
1109 if (Processor::saveState(pCurrentThread->state())) {
1110 // Just context-restored.
1111 if (bWasInterrupts)
1113 return;
1114 }
1115
1116 if (bUsermode) {
1118 reinterpret_cast<uintptr_t>(pStartFunction),
1119 reinterpret_cast<uintptr_t>(pStack), reinterpret_cast<uintptr_t>(pParam));
1120 } else {
1122 reinterpret_cast<uintptr_t>(pStartFunction),
1123 reinterpret_cast<uintptr_t>(pStack),
1124 reinterpret_cast<uintptr_t>(pParam));
1125 }
1126 }
1127}
1128
1129void PerProcessorScheduler::addThread(Thread* pThread, SyscallState& state) {
1130 // Handle wrong CPU, and handle thread not yet ready to schedule.
1131 if (this != &Processor::information().getScheduler() || pThread->getStatus() == Thread::Created) {
1132 newThreadData* pData = new newThreadData;
1133 pData->pThread = pThread;
1134 pData->pParam = nullptr;
1135 pData->useSyscallState = true;
1136 pData->state = state;
1137
1138 pThread->m_Lock.release();
1139
1140 m_NewThreadDataLock.acquire();
1141 if (!m_NewThreadAdmissionOpen) {
1142 m_NewThreadDataLock.release();
1143 delete pData;
1144 panic("Thread admitted after its per-processor worker stopped.");
1145 }
1146 m_NewThreadData.pushBack(pData);
1147 m_NewThreadDataLock.release();
1148
1149 m_NewThreadDataCondition.signal();
1150 return;
1151 }
1152
1153 pThread->setCpuId(Processor::id());
1154 pThread->setScheduler(this);
1155
1156 bool bWasInterrupts = Processor::getInterrupts();
1158
1159 // We assume here that pThread's lock is already taken.
1160
1161 Thread* pCurrentThread = Processor::information().getCurrentThread();
1162
1163 // Grab the current thread's lock.
1164 pCurrentThread->getLock().acquire();
1165
1167
1168 assert(pThread->m_Placement.allowed.contains(m_LogicalCpu));
1169 pThread->m_HasSchedulerContext = true;
1170 // Now neither thread can be moved, we're safe to switch.
1171 Metrics::increment(Metrics::Counter::ContextSwitch);
1172 if (pThread == m_pIdleThread) {
1173 Metrics::increment(Metrics::Counter::IdleSelection);
1174 }
1175
1176 if (pCurrentThread != m_pIdleThread) {
1177 pCurrentThread->setStatusUnlocked(Thread::Ready);
1178 }
1179 pThread->setStatusUnlocked(Thread::Running);
1180 Processor::information().setCurrentThread(pThread);
1181 updateOneShotTimer();
1182 void* kernelStack = pThread->getKernelStack();
1183 Processor::information().setKernelStack(reinterpret_cast<uintptr_t>(kernelStack));
1186
1187 // This thread is safe from being moved as its status is now "running".
1188 // It is worth noting that we can't just call exit() here, as the lock is
1189 // not necessarily actually taken.
1190 if (pThread->getLock().interrupts())
1191 bWasInterrupts = true;
1192 bool bWas = pThread->getLock().acquired();
1193 pThread->getLock().unlockForScheduler();
1194 EMIT_IF(TRACK_LOCKS) {
1195 g_LocksCommand.lockReleased(&pCurrentThread->getLock());
1196 if (bWas) {
1197 // We unlocked the lock, so track that unlock.
1198 g_LocksCommand.lockReleased(&pThread->getLock());
1199 }
1200 if (!g_LocksCommand.checkSchedule()) {
1201 FATAL("Lock checker disallowed this reschedule.");
1202 }
1203 }
1204
1205#if X64 && !HOSTED
1206 // CLONE_SETTLS may have replaced the copied parent's base above.
1207 state.refreshUserTlsBase();
1208#endif
1209
1210 // Copy the SyscallState into this thread's kernel stack.
1211 uintptr_t kStack = reinterpret_cast<uintptr_t>(pThread->getKernelStack());
1212 kStack -= sizeof(SyscallState);
1213 MemoryCopy(reinterpret_cast<void*>(kStack), reinterpret_cast<void*>(&state),
1214 sizeof(SyscallState));
1215
1216 // Grab a reference to the stack in the form of a full SyscallState.
1217 SyscallState& newState = *reinterpret_cast<SyscallState*>(kStack);
1218
1219 pCurrentThread->trackTime(CpuTimeMode::Kernel);
1220 // restoreState(SyscallState) returns directly to the forked userspace
1221 // frame, so establish its user baseline before the no-return handoff.
1222 pThread->recordTime(CpuTimeMode::User);
1223
1224 EMIT_IF(SYSTEM_REQUIRES_ATOMIC_CONTEXT_SWITCH) {
1225 NOTICE("restoring (new) syscall state");
1226 Processor::switchState(bWasInterrupts, pCurrentThread->state(), newState,
1227 pCurrentThread->getLock().deferredReleaseWord());
1228 }
1229 else {
1230 if (Processor::saveState(pCurrentThread->state())) {
1231 // Just context-restored.
1232 if (bWasInterrupts)
1234 return;
1235 }
1236
1237 Processor::restoreState(newState, pCurrentThread->getLock().deferredReleaseWord());
1238 }
1239}
1240
1242 Thread* pThread = Processor::information().getCurrentThread();
1243 if (!pThread) {
1244 FATAL("Clean thread exit has no current Thread.");
1245 }
1246
1247 if (__atomic_load_n(&pThread->m_TerminationDeferralDepth, __ATOMIC_ACQUIRE) ||
1248 __atomic_load_n(&pThread->m_EventDeferralDepth, __ATOMIC_ACQUIRE)) {
1249 FATAL("Clean thread exit reached a boundary with active deferral scopes.");
1250 }
1251
1252 for (size_t level = 0; level < MAX_NESTED_EVENTS; ++level) {
1253 if (__atomic_load_n(&pThread->m_pDeferredScopes[level], __ATOMIC_ACQUIRE)) {
1254 FATAL("Clean thread exit reached a boundary with armed cleanup records.");
1255 }
1256 }
1257 if (pThread->hasActiveWaitAtAnyLevel()) {
1258 FATAL("Clean thread exit reached a boundary with an active WaitQueue record.");
1259 }
1260
1261 const bool transferToIdle = pThread && pThread->m_ExitToIdle;
1262 pThread->m_ExitToIdle = false;
1263 finishCurrentThreadExit(pLock, transferToIdle);
1264}
1265
1267 abandonCurrentThreadStack(StackDiscardReason::LegacyAbiCall, pLock);
1268}
1269
1271 Thread* pThread = Processor::information().getCurrentThread();
1272 if (!pThread) {
1273 FATAL("Stack discard has no current Thread.");
1274 }
1275
1276 g_StackDiscardCount += 1;
1277 switch (reason) {
1278 case StackDiscardReason::EmergencyProcessKill:
1279 g_EmergencyProcessKillDiscardCount += 1;
1280 break;
1281 case StackDiscardReason::HostedRegression:
1282 g_HostedRegressionDiscardCount += 1;
1283 break;
1284 case StackDiscardReason::LegacyAbiCall:
1285 g_LegacyAbiDiscardCount += 1;
1286 break;
1287 }
1288
1289 // Cleanup callbacks can destroy the object which owns a published wait
1290 // queue. Unlink intrusive wait records before invoking those callbacks.
1291 pThread->unlinkWaitsForStackDiscard();
1292
1293 // No C++ destructors run after this call. Retire registered stack-owned
1294 // lifetime records while their abandoned storage is still mapped.
1295 pThread->retireDeferredScopes(true);
1296
1297 const bool transferToIdle = pThread->m_ExitToIdle;
1298 pThread->m_ExitToIdle = false;
1299 finishCurrentThreadExit(pLock, transferToIdle);
1300}
1301
1303 return g_StackDiscardCount.value();
1304}
1305
1307 switch (reason) {
1308 case StackDiscardReason::EmergencyProcessKill:
1309 return g_EmergencyProcessKillDiscardCount.value();
1310 case StackDiscardReason::HostedRegression:
1311 return g_HostedRegressionDiscardCount.value();
1312 case StackDiscardReason::LegacyAbiCall:
1313 return g_LegacyAbiDiscardCount.value();
1314 }
1315 return 0;
1316}
1317
1319 Thread* pThread = Processor::information().getCurrentThread();
1320 if (!pThread || !m_pIdleThread || pThread == m_pIdleThread) {
1321 panic("Current thread has no distinct idle shutdown owner!");
1322 }
1323 pThread->m_ExitToIdle = true;
1324}
1325
1326void PerProcessorScheduler::finishCurrentThreadExit(Spinlock* pLock, bool transferToIdle) {
1327 Thread* pThread = Processor::information().getCurrentThread();
1328
1329 // Start shutting down the current thread while we can still schedule it.
1330 pThread->shutdown();
1331
1333 PerProcessorScheduler& owner = Processor::information().getScheduler();
1334
1335 // Removing the current thread. Grab its lock.
1336 pThread->getLock().acquire();
1337
1338 // If we're tracking locks, don't pollute the results. Yes, we've kept
1339 // this lock held, but it no longer matters.
1340 EMIT_IF(TRACK_LOCKS) {
1341 g_LocksCommand.lockReleased(&pThread->getLock());
1342 if (pLock) {
1343 g_LocksCommand.lockReleased(pLock);
1344 }
1345 if (!g_LocksCommand.checkSchedule()) {
1346 FATAL("Lock checker disallowed this reschedule.");
1347 }
1348 }
1349
1350 // Get another thread ready to schedule.
1351 // This will also get the lock for the returned thread.
1352 Thread* pNextThread = transferToIdle ? nullptr : owner.selectNext(pThread);
1353
1354 if (transferToIdle && (!owner.m_pIdleThread || owner.m_pIdleThread == pThread)) {
1355 panic("Current thread has no distinct idle shutdown owner!");
1356 }
1357
1358 if (pNextThread == 0 && owner.m_pIdleThread == 0) {
1359 // Nothing to switch to, we're in a VERY bad situation.
1360 panic("Attempting to kill only thread on this processor!");
1361 } else if (pNextThread == 0) {
1362 pNextThread = owner.m_pIdleThread;
1363 if (pNextThread != pThread)
1364 pNextThread->getLock().acquire();
1365 }
1366
1367 Metrics::increment(Metrics::Counter::ContextSwitch);
1368 if (pNextThread == owner.m_pIdleThread) {
1369 Metrics::increment(Metrics::Counter::IdleSelection);
1370 }
1371 pNextThread->setStatusUnlocked(Thread::Running);
1372 Processor::information().setCurrentThread(pNextThread);
1373 owner.updateOneShotTimer();
1374 void* kernelStack = pNextThread->getKernelStack();
1375 Processor::information().setKernelStack(reinterpret_cast<uintptr_t>(kernelStack));
1376 EMIT_IF(!HOSTED) {
1378 }
1379 Processor::setTlsBase(pNextThread->getTlsBase());
1380
1381 pThread->trackTime(CpuTimeMode::Kernel);
1382 pNextThread->recordTime(CpuTimeMode::Kernel);
1383
1384 pNextThread->getLock().exit();
1385
1386 // Pass in the lock atom we were given if possible, as the caller wants an
1387 // atomic release (i.e. once the thread is no longer able to be scheduled).
1388 deleteThreadThenRestoreState(pThread, pNextThread->state(),
1389 pLock ? pLock->deferredReleaseWord() : 0);
1390}
1391
1392void PerProcessorScheduler::deleteThread(Thread* pThread) {
1393#if HOSTED
1394 // A hosted user Thread is still ASan's active fiber until the assembly
1395 // handoff has moved onto safe_stack. Unmapping its address space earlier
1396 // makes ASan—and potentially host signal machinery—observe a live stack
1397 // disappearing underneath the no-return transition.
1398 Thread* replacement = Processor::information().getCurrentThread();
1399 if (!replacement || replacement == pThread) {
1400 FATAL("Hosted Thread deletion has no replacement address space");
1401 }
1403#endif
1404
1405 Process* pProcess = pThread->getParent();
1406 // This runs on a temporary handoff stack before the replacement Thread
1407 // has retired any WaitQueue record it resumed from. Blocking here would
1408 // try to enrol that Thread in two queues at once. Close admission now;
1409 // joins drain in ordinary thread context, while the final external lease
1410 // completes deferred detached deletion.
1411 pThread->closeExternalLeaseAdmission();
1412
1413 // This is the scheduler's final access outside the Process-serialised
1414 // retirement block below. Publish the unlocked atom before m_bReapable so
1415 // a joiner or detach owner can treat reapable as a true final-use boundary.
1416 // No Process lock is held here, so waking another same-core thread cannot
1417 // resume it into a conflicting terminal operation under that lock.
1418 pThread->getLock().unlockForScheduler();
1419
1420 bool deleteTarget = false;
1421 bool completesProcessExit = false;
1422 bool wakeExitOwner = false;
1423 {
1424 RecursingLockGuard<Spinlock> processGuard(pProcess->m_Lock);
1425 deleteTarget = pThread->markReapable();
1426 completesProcessExit = pProcess->terminatingThreadReapable(pThread, wakeExitOwner);
1427
1428 if (deleteTarget) {
1429 if (!pProcess->m_DeferredThreadReaps.tryEnter()) {
1430 FATAL_NOLOCK("Thread retirement raced closed Process admission.");
1431 }
1432 }
1433 }
1434
1435 // Process-exit progress is a predicate update under m_Lock followed by an
1436 // out-of-lock notification. Keeping the WaitQueue wake outside m_Lock
1437 // prevents a resumed owner from acquiring the queue under an outer lock.
1438 if (wakeExitOwner) {
1439 pProcess->m_TerminationWaiters.wakeAll();
1440 pProcess->m_ExecWaiters.wakeAll();
1441 }
1442
1443 if (deleteTarget) {
1444 PerProcessorScheduler& scheduler = Processor::information().getScheduler();
1445 if (pThread->getScheduler() && pThread->getScheduler() != &scheduler) {
1446 FATAL_NOLOCK("Thread retirement reached a non-owning scheduler.");
1447 }
1448 scheduler.publishDeferredThreadReap(pThread);
1449 }
1450
1451 if (!completesProcessExit) {
1452 return;
1453 }
1454
1455 // This is the final Process access: publication can wake a reaper that
1456 // destroys both the Process and its retained, reapable Thread objects.
1457 pProcess->publishTerminationReapable();
1458}
1459
1460void PerProcessorScheduler::removeThread(Thread* pThread) {
1462}
1463
1465 schedule(Thread::Sleeping);
1466}
1467
1469 if (!pThread) {
1470 assert(false);
1471 return;
1472 }
1473 pThread->publishReadyNotification();
1474}
1475
1476Thread* PerProcessorScheduler::selectNext(Thread* current) {
1477 // Keep selecting the idle owner until it retires that role itself. A tick
1478 // can preempt its first resumed turn before it observes the shutdown flag.
1479 if (__atomic_load_n(&m_IdleWakeRequested, __ATOMIC_ACQUIRE)) {
1480 Thread* idle = __atomic_load_n(&m_pIdleThread, __ATOMIC_ACQUIRE);
1481 if (idle) {
1482 if (idle == current)
1483 return nullptr;
1484 idle->m_Lock.acquire();
1485 return idle;
1486 }
1487 }
1488 while (Thread* candidate = m_pSchedulingAlgorithm->getNext(current)) {
1489 candidate->m_Lock.acquire();
1490 if (candidate->getScheduler() == this && candidate->m_Status == Thread::Ready &&
1491 !candidate->m_ReadyPublicationPending)
1492 return candidate;
1493 candidate->m_Lock.release();
1494 }
1495 return nullptr;
1496}
1497
1498void PerProcessorScheduler::timer(uint64_t delta, InterruptState& state) {
1499 (void)state;
1500 if constexpr (PEDIGREE_TIME_ACCOUNTING && PEDIGREE_SAMPLED_TIME_ACCOUNTING) {
1501 Thread* current = Processor::information().getCurrentThread();
1502 if (current && current != m_pIdleThread && delta) {
1503 current->accountTimerTick(delta, state.kernelMode());
1504 }
1505 }
1506 ActivityDiagnostics::recordSchedulerTimer();
1507 Metrics::increment(Metrics::Counter::Timer);
1508 if (m_OneShotTimer) {
1509 if (!delta) {
1510 programOneShotTimer();
1511 return;
1512 }
1513 const uint64_t now = Time::getTicksFast();
1514 if (m_NextLoadSampleDeadline && now >= m_NextLoadSampleDeadline) {
1515 m_NextLoadSampleDeadline = now > ~uint64_t(0) - LoadAverage::PeriodNanoseconds
1516 ? ~uint64_t(0)
1517 : now + LoadAverage::PeriodNanoseconds;
1519 }
1520 if (m_QuantumDeadline && now >= m_QuantumDeadline) {
1521 m_QuantumDeadline = 0;
1522 m_ReschedulePending = 1;
1523 }
1524 const uint64_t clockDeadline = m_ClockDeadline.value();
1525 if (clockDeadline && now >= clockDeadline && m_ClockDeadline.compareAndSwap(clockDeadline, 0)) {
1526 Machine::instance().getTimer()->deadlineInterrupt();
1527 }
1528 programOneShotTimer();
1529 return;
1530 }
1531 if (delta) {
1533 }
1534 // The raw handler only records the scheduling edge. The architecture IRQ
1535 // return services it after all hard-handler scopes have unwound.
1536 if (!delta) {
1537 m_ReschedulePending = 1;
1538 } else if (++m_SchedulerTickCounter >= PEDIGREE_SCHEDULER_TICK_DIVISOR) {
1539 m_SchedulerTickCounter = 0;
1540 m_ReschedulePending = 1;
1541 }
1542}
1543
1544void PerProcessorScheduler::threadStatusChanged(Thread* pThread) {
1545 bool wakeWorker = false;
1546 bool queueLocked = false;
1547 PerProcessorScheduler* readyOwner = nullptr;
1548 pThread->m_Lock.acquire();
1549 if (pThread->m_Status == Thread::Created) {
1550 // Acquiring the queue mutex may schedule its worker on this CPU. Never
1551 // retain the thread spinlock that the worker needs while waiting for it.
1552 pThread->m_Lock.release();
1553 m_NewThreadDataLock.acquire();
1554 queueLocked = true;
1555 pThread->m_Lock.acquire();
1556 if (pThread->m_Status == Thread::Created) {
1557 for (List<void*>::Iterator it = m_DelayedNewThreadData.begin();
1558 it != m_DelayedNewThreadData.end();) {
1559 newThreadData* pData = reinterpret_cast<newThreadData*>(*it);
1560 if (pData->pThread == pThread) {
1561 void* p = *it;
1562 it = m_DelayedNewThreadData.erase(it);
1563 m_NewThreadData.pushBack(p);
1564 wakeWorker = true;
1565 } else {
1566 ++it;
1567 }
1568 }
1569 }
1570 }
1571
1572 PerProcessorScheduler* owner = pThread->getScheduler();
1573 assert(owner);
1574 owner->m_pSchedulingAlgorithm->threadStatusChanged(pThread);
1575 if (pThread->m_Status == Thread::Ready && !pThread->m_ReadyPublicationPending) {
1576 readyOwner = owner;
1577 }
1578 pThread->m_Lock.release();
1579 if (queueLocked) {
1580 m_NewThreadDataLock.release();
1581 }
1582
1583 if (wakeWorker) {
1584 m_NewThreadDataCondition.signal();
1585 }
1586 if (readyOwner) {
1587 readyOwner->prompt(true);
1588 }
1589}
1590
1592 m_IrqWorkDoorbell = 1;
1593 prompt();
1594}
1595
1597 LockGuard<Spinlock> guard(m_IrqWorkLock);
1598 if (worker.m_pWaiters || worker.m_pNext) {
1599 FATAL("Scheduler worker wake was registered twice.");
1600 }
1601 worker.m_Pending = 0;
1602 worker.m_pWaiters = &waiters;
1603 worker.m_pNext = m_pWorkerWakeHead;
1604 m_pWorkerWakeHead = &worker;
1605}
1606
1608 LockGuard<Spinlock> guard(m_IrqWorkLock);
1609 SchedulerWorkerWake** link = &m_pWorkerWakeHead;
1610 while (*link && *link != &worker) {
1611 link = &(*link)->m_pNext;
1612 }
1613 if (!*link) {
1614 FATAL("Scheduler worker wake was not registered.");
1615 }
1616 *link = worker.m_pNext;
1617 worker.m_pNext = nullptr;
1618 worker.m_pWaiters = nullptr;
1619 worker.m_Pending = 0;
1620}
1621
1623 worker.m_Pending = 1;
1624 if (promptOwner) {
1626 } else {
1627 m_IrqWorkDoorbell = 1;
1628 }
1629}
1630
1631void PerProcessorScheduler::serviceWorkerWakeups() {
1632 LockGuard<Spinlock> guard(m_IrqWorkLock);
1633 bool retry = false;
1634 for (SchedulerWorkerWake* worker = m_pWorkerWakeHead; worker; worker = worker->m_pNext) {
1635 if (!worker->m_Pending.value() || !worker->m_pWaiters) {
1636 continue;
1637 }
1638
1639 if (worker->m_pWaiters->wakeOne()) {
1640 Metrics::increment(Metrics::Counter::WorkerWake);
1641 worker->m_Pending.compareAndSwap(1, 0);
1642 } else {
1643 // A producer may publish between the worker's empty check and its
1644 // WaitQueue enrollment. Keep the edge armed so the next scheduler
1645 // boundary retries after the waiter is visible.
1646 retry = true;
1647 }
1648 }
1649 if (retry) {
1650 m_IrqWorkDoorbell = 1;
1651 }
1652}
1653
1655 m_TimeAccountingState.publish();
1656 ringIrqWorkDoorbell(m_TimeAccountingWorkerWake);
1657}
1658
1660 Metrics::increment(Metrics::Counter::RescheduleService);
1661 if (!m_pSchedulingAlgorithm || !Processor::information().getCurrentThread()) {
1662 return;
1663 }
1664
1665 // One bounded claim is enough: a racing ring remains set for the next
1666 // scheduler tick.
1667 if (m_IrqWorkDoorbell.compareAndSwap(1, 0)) {
1668 serviceWorkerWakeups();
1669 m_ReschedulePending.compareAndSwap(1, 0);
1670 schedule();
1671 }
1672}
1673
1675 Metrics::increment(Metrics::Counter::RescheduleService);
1676 if (!m_pSchedulingAlgorithm || !Processor::information().getCurrentThread()) {
1677 return;
1678 }
1679
1680 m_RemotePromptPending = 0;
1681 bool pending = m_ReschedulePending.compareAndSwap(1, 0);
1682 if (m_IrqWorkDoorbell.compareAndSwap(1, 0)) {
1683 serviceWorkerWakeups();
1684 pending = true;
1685 m_ReschedulePending.compareAndSwap(1, 0);
1686 }
1687 if (pending) {
1688 schedule(Thread::Ready, false);
1689 }
1690}
1691
1694 Processor::executionContext() != ExecutionContext::WaitableThread) {
1695 FATAL_NOLOCK(
1696 "Return-to-user stop work requires an IRQ-enabled thread "
1697 "boundary.");
1698 }
1699
1700 Thread* current = Processor::information().getCurrentThread();
1701 if (current && current->currentTimeAccountingMode() != CpuTimeMode::Kernel) {
1702 FATAL_NOLOCK("Return-to-user stop work escaped Kernel accounting mode");
1703 }
1704 if (!current) {
1705 return false;
1706 }
1707
1708 Process* process = current->getParent();
1709 if (!process) {
1710 return current->getUnwindState() != Thread::Continue;
1711 }
1712
1713 while (true) {
1714 bool dispatchKernelEvent = false;
1715 Thread::EventSelection selection = Thread::EventSelection::StoppedProcessKernel;
1716 WaitQueue::WakeReason reason = WaitQueue::WakeReason::Spurious;
1717 {
1718 // Signal publication and resume both use this queue, making the
1719 // predicate check and waiter publication one atomic handshake.
1720 auto guard = current->m_EventWaiters.acquire();
1721 {
1722 LockGuard<Spinlock> threadGuard(current->m_Lock);
1723 if (current->getUnwindState() != Thread::Continue) {
1724 return true;
1725 }
1726
1727 const Process::ProcessState state = process->getState();
1728 if (state == Process::Active) {
1729 selection = Thread::EventSelection::KernelDeliverable;
1730 dispatchKernelEvent = mode == ProcessStopGateMode::DirectUserTransition &&
1731 current->hasDeliverableEventsUnlocked(selection);
1732 if (!dispatchKernelEvent) {
1733 return false;
1734 }
1735 }
1736 if (state == Process::Terminated || state == Process::Reaped) {
1737 FATAL_NOLOCK("A live Thread reached userspace after its Process terminated");
1738 }
1739
1740 // SIGKILL and other kernel events that explicitly permit stopped
1741 // delivery must run before this thread can wait indefinitely.
1742 if (state == Process::Suspended) {
1743 selection = Thread::EventSelection::StoppedProcessKernel;
1744 dispatchKernelEvent = current->hasDeliverableEventsUnlocked(selection);
1745 }
1746 }
1747
1748 if (!dispatchKernelEvent) {
1749 // A Terminating peer can observe the Process transition just before
1750 // beginTermination publishes that peer's unwind state. Its terminal
1751 // wake releases this same wait.
1752 reason = guard.waitWithoutEventDispatch(
1753 WaitQueue::Channel(process, static_cast<uintptr_t>(Thread::ProcessWait)),
1754 Thread::ProcessWait, reinterpret_cast<uintptr_t>(__builtin_return_address(0)));
1755 }
1756 }
1757
1758 if (dispatchKernelEvent) {
1759 // Direct transitions have no reusable user stack. A selection-specific
1760 // dequeue prevents a racing resume from broadening stopped delivery.
1761 Processor::information().getScheduler().checkEventState(0, selection);
1762 continue;
1763 }
1764 if (reason == WaitQueue::WakeReason::Terminating ||
1765 reason == WaitQueue::WakeReason::Unwinding) {
1766 return true;
1767 }
1768 }
1769}
1770
1772 UserReturnFrame::Origin origin,
1773 bool diagnosticSample) {
1774 const uint64_t workStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1775 auto finishWork = [diagnosticSample, workStart](bool terminal) {
1776 if (diagnosticSample) {
1777 ActivityDiagnostics::recordUserReturnStage(
1778 ActivityDiagnostics::UserReturnStage::InterruptWork,
1779 ActivityDiagnostics::timestamp() - workStart);
1780 }
1781 return terminal;
1782 };
1783 Thread* owner = Processor::information().getCurrentThread();
1784 if (!owner)
1785 return finishWork(false);
1786#if X64 && !HOSTED
1787 {
1788 EnsureInterrupts interrupts(false);
1789 state.setFlags(state.getFlags() | 0x202);
1790 }
1791#endif
1792#if PEDIGREE_FAST_USER_RETURN
1793 if ((origin == UserReturnFrame::Origin::Syscall ||
1794 origin == UserReturnFrame::Origin::Interrupt) &&
1795 owner->canSkipUserReturnWork()) {
1796 return finishWork(false);
1797 }
1798#endif
1799 UserReturnFrame frame(*owner, state, origin);
1800 Thread::UserReturnFrameScope frameScope(*owner, frame);
1801
1802 Subsystem* subsystem = owner->getParent() ? owner->getParent()->getSubsystem() : nullptr;
1803 if (subsystem) {
1804 const uint64_t checkpointStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1805 const bool terminal =
1806 subsystem->userReturnCheckpoint(*owner, frame) == Subsystem::UserReturnResult::Terminal;
1807 if (diagnosticSample) {
1808 ActivityDiagnostics::recordUserReturnStage(
1809 ActivityDiagnostics::UserReturnStage::Checkpoint,
1810 ActivityDiagnostics::timestamp() - checkpointStart);
1811 }
1812 if (terminal)
1813 return finishWork(true);
1814 }
1815
1816 // Terminal requests and process stops win over later work. The architecture
1817 // caller owns the final commit after its return-tail scopes and accounting
1818 // have retired.
1819 uint64_t stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1820 bool terminal = Processor::information().getScheduler().serviceProcessStopAtUserReturn();
1821 if (diagnosticSample) {
1822 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::ProcessStop,
1823 ActivityDiagnostics::timestamp() - stageStart);
1824 }
1825 if (terminal)
1826 return finishWork(true);
1827
1828 Processor::information().getScheduler().serviceDeferredSubsystemException(state,
1829 diagnosticSample);
1830 Thread* current = Processor::information().getCurrentThread();
1831 if (current && !current->isTerminationDeferred() &&
1832 current->getUnwindState() != Thread::Continue) {
1833 return finishWork(true);
1834 }
1835
1836#if PEDIGREE_FAST_USER_RETURN
1837 // Resolving the deferred fault may have retired the only pending work.
1838 if ((origin == UserReturnFrame::Origin::Syscall ||
1839 origin == UserReturnFrame::Origin::Interrupt) &&
1840 !frame.m_Terminal && owner->canSkipUserReturnWork()) {
1841 return finishWork(false);
1842 }
1843#endif
1844
1845 stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1846 terminal = Processor::information().getScheduler().serviceProcessStopAtUserReturn();
1847 if (diagnosticSample) {
1848 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::ProcessStop,
1849 ActivityDiagnostics::timestamp() - stageStart);
1850 }
1851 if (terminal)
1852 return finishWork(true);
1853
1854 stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1855 Processor::information().getScheduler().checkEventState(
1856 state.getStackPointer(), Thread::EventSelection::AnyDeliverable, &state, nullptr);
1857 if (diagnosticSample) {
1858 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::Event,
1859 ActivityDiagnostics::timestamp() - stageStart);
1860 }
1861
1862 stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1863 terminal =
1864 frame.m_Terminal || Processor::information().getScheduler().serviceProcessStopAtUserReturn();
1865 if (diagnosticSample) {
1866 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::ProcessStop,
1867 ActivityDiagnostics::timestamp() - stageStart);
1868 }
1869 if (!terminal)
1870 owner->clearUserReturnWorkIfIdle();
1871 return finishWork(terminal);
1872}
1873
1875 UserReturnFrame::Origin origin,
1876 bool diagnosticSample) {
1877 const uint64_t workStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1878 auto finishWork = [diagnosticSample, workStart](bool terminal) {
1879 if (diagnosticSample) {
1880 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::SyscallWork,
1881 ActivityDiagnostics::timestamp() - workStart);
1882 }
1883 return terminal;
1884 };
1885 Thread* owner = Processor::information().getCurrentThread();
1886 if (!owner)
1887 return finishWork(false);
1888#if X64 && !HOSTED
1889 {
1890 EnsureInterrupts interrupts(false);
1891 state.setFlags(state.getFlags() | 0x202);
1892 }
1893#endif
1894#if PEDIGREE_FAST_USER_RETURN
1895 if (origin == UserReturnFrame::Origin::Syscall && owner->canSkipUserReturnWork())
1896 return finishWork(false);
1897#endif
1898
1899 UserReturnFrame frame(*owner, state, origin);
1900 Thread::UserReturnFrameScope frameScope(*owner, frame);
1901 Subsystem* subsystem = owner->getParent() ? owner->getParent()->getSubsystem() : nullptr;
1902 if (subsystem) {
1903 const uint64_t checkpointStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1904 const bool terminal =
1905 subsystem->userReturnCheckpoint(*owner, frame) == Subsystem::UserReturnResult::Terminal;
1906 if (diagnosticSample) {
1907 ActivityDiagnostics::recordUserReturnStage(
1908 ActivityDiagnostics::UserReturnStage::Checkpoint,
1909 ActivityDiagnostics::timestamp() - checkpointStart);
1910 }
1911 if (terminal)
1912 return finishWork(true);
1913 }
1914 uint64_t stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1915 bool terminal = Processor::information().getScheduler().serviceProcessStopAtUserReturn();
1916 if (diagnosticSample) {
1917 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::ProcessStop,
1918 ActivityDiagnostics::timestamp() - stageStart);
1919 }
1920 if (terminal)
1921 return finishWork(true);
1922 stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1923 Processor::information().getScheduler().checkEventState(
1924 state.getStackPointer(), Thread::EventSelection::AnyDeliverable, nullptr, &state);
1925 if (diagnosticSample) {
1926 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::Event,
1927 ActivityDiagnostics::timestamp() - stageStart);
1928 }
1929 stageStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1930 terminal =
1931 frame.m_Terminal || Processor::information().getScheduler().serviceProcessStopAtUserReturn();
1932 if (diagnosticSample) {
1933 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::ProcessStop,
1934 ActivityDiagnostics::timestamp() - stageStart);
1935 }
1936 if (!terminal)
1937 owner->clearUserReturnWorkIfIdle();
1938 return finishWork(terminal);
1939}
1940
1942 bool diagnosticSample) {
1943 Thread* thread = Processor::information().getCurrentThread();
1944 if (!thread) {
1945 return;
1946 }
1947
1948 size_t rawType = 0;
1949 uintptr_t faultAddress = 0;
1950 uintptr_t errorCode = 0;
1951 if (!thread->takeDeferredSubsystemException(rawType, faultAddress, errorCode)) {
1952 return;
1953 }
1954
1955 Process* process = thread->getParent();
1956 Subsystem* subsystem = process ? process->getSubsystem() : nullptr;
1957 if (!subsystem || rawType > static_cast<size_t>(Subsystem::Other)) {
1958 FATAL("Deferred userspace exception has no valid owning subsystem");
1959 }
1960
1961 // Disk-backed faults can only wait once the raw interrupt and accounting
1962 // scopes are gone. An unsuccessful retry retains ordinary signal delivery.
1963 if (rawType == static_cast<size_t>(Subsystem::PageFault) && !state.kernelMode() &&
1965 const uint64_t faultStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
1966 const bool handled = subsystem->resolveUserPageFault(*thread, state, faultAddress, errorCode);
1967 if (diagnosticSample) {
1968 ActivityDiagnostics::recordUserReturnStage(
1969 ActivityDiagnostics::UserReturnStage::DeferredFault,
1970 ActivityDiagnostics::timestamp() - faultStart);
1971 ActivityDiagnostics::recordUserReturnFaultOutcome(handled);
1972 }
1973 if (handled)
1974 return;
1975 }
1976
1977 subsystem->threadException(thread, static_cast<Subsystem::ExceptionType>(rawType), &state,
1978 faultAddress, errorCode);
1979}
1980
1982 Thread* thread = Processor::information().getCurrentThread();
1983 if (!thread || thread->isTerminationDeferred()) {
1984 return;
1985 }
1986
1987 const Thread::UnwindType unwindState = thread->getUnwindState();
1988 if (unwindState == Thread::TerminateThread) {
1990 }
1991 if (unwindState == Thread::Exit) {
1992 Process* process = thread->getParent();
1993 Subsystem* subsystem = process ? process->getSubsystem() : nullptr;
1994 if (!subsystem) {
1995 FATAL("Return-to-user process exit has no owning subsystem.");
1996 }
1997 const Thread::DeferredProcessExit request = thread->takeDeferredProcessExit();
1998 subsystem->exit(request.code, request.cause);
1999 FATAL("Subsystem::exit returned to a user-return boundary.");
2000 }
2001}
2002
2003#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2004namespace {
2005struct HostedRunnableCurrentContext {
2006 HostedRunnableCurrentContext(Thread* current, Thread* idleOwner)
2007 : driver(current), idle(idleOwner), idleWhileRunnable(0) {}
2008
2009 Thread* driver;
2010 Thread* idle;
2011 Atomic<size_t> idleWhileRunnable;
2012};
2013
2014HostedRunnableCurrentContext* g_HostedRunnableCurrentContext = nullptr;
2015
2016void observeHostedRunnableCurrent(ProcessorBase::HostedContextSwitchStage stage) {
2017 auto* context = __atomic_load_n(&g_HostedRunnableCurrentContext, __ATOMIC_ACQUIRE);
2018 if (context && stage == ProcessorBase::HostedContextSwitchStage::SwitchStateReturnedMasked &&
2019 Processor::information().getCurrentThread() == context->idle &&
2020 context->driver->getStatus() == Thread::Ready) {
2021 context->idleWhileRunnable += 1;
2022 }
2023}
2024
2025struct HostedNewThreadContext {
2026 Atomic<size_t> calls;
2027};
2028
2029int hostedNewThreadWorkerEntry(void* parameter) {
2030 HostedNewThreadContext* context = reinterpret_cast<HostedNewThreadContext*>(parameter);
2031 context->calls += 1;
2032 return 0;
2033}
2034
2035} // namespace
2036
2037bool PerProcessorScheduler::currentIrqWorkDoorbellPendingForTest() {
2038 return Processor::information().getScheduler().m_IrqWorkDoorbell.value() != 0;
2039}
2040
2041void PerProcessorScheduler::serviceCurrentIrqWorkDoorbellForTest() {
2042 Processor::information().getScheduler().serviceIrqWorkDoorbell();
2043}
2044
2045bool PerProcessorScheduler::runHostedNewThreadWorkerRegressions() {
2046 if (this != &Processor::information().getScheduler() || !m_NewThreadWorker) {
2047 ERROR(
2048 "HOSTED-WAIT-TEST: per-processor worker regression requires "
2049 "the active scheduler");
2050 return false;
2051 }
2052
2053 constexpr size_t Attempts = 10000;
2054 auto isParked = [this](Thread* target) {
2055 bool found = false;
2056 m_NewThreadDataLock.acquire();
2057 for (List<void*>::Iterator it = m_DelayedNewThreadData.begin();
2058 it != m_DelayedNewThreadData.end(); ++it) {
2059 newThreadData* pData = reinterpret_cast<newThreadData*>(*it);
2060 if (pData->pThread == target) {
2061 found = true;
2062 break;
2063 }
2064 }
2065 m_NewThreadDataLock.release();
2066 return found;
2067 };
2068 auto waitUntilParked = [&isParked, Attempts](Thread* target) {
2069 for (size_t attempt = 0; attempt < Attempts; ++attempt) {
2070 if (isParked(target)) {
2071 return true;
2072 }
2074 }
2075 return false;
2076 };
2077 auto check = [](bool condition, const char* message) {
2078 if (!condition) {
2079 ERROR("HOSTED-WAIT-TEST: " << message);
2080 }
2081 return condition;
2082 };
2083
2084 Process* kernelProcess = Processor::information().getCurrentThread()->getParent();
2085 bool passed = true;
2086
2087 HostedRunnableCurrentContext runnableContext(Processor::information().getCurrentThread(),
2088 m_pIdleThread);
2089 if (!check(runnableContext.idle && runnableContext.idle != runnableContext.driver,
2090 "runnable-current regression requires an ordinary thread and an idle owner")) {
2091 return false;
2092 }
2093 __atomic_store_n(&g_HostedRunnableCurrentContext, &runnableContext, __ATOMIC_RELEASE);
2094 Processor::setHostedContextSwitchHook(observeHostedRunnableCurrent);
2095 for (size_t attempt = 0; attempt < 256; ++attempt) {
2097 }
2098 Processor::setHostedContextSwitchHook(nullptr);
2099 __atomic_store_n(&g_HostedRunnableCurrentContext,
2100 static_cast<HostedRunnableCurrentContext*>(nullptr), __ATOMIC_RELEASE);
2101 const bool runnablePassed =
2102 check(!runnableContext.idleWhileRunnable && Processor::getInterrupts() &&
2103 Processor::information().getCurrentThread() == runnableContext.driver &&
2104 runnableContext.driver->getStatus() == Thread::Running,
2105 "scheduler entered idle while the yielding thread remained runnable");
2106 passed &= runnablePassed;
2107 if (runnablePassed) {
2108 NOTICE("HOSTED-WAIT-TEST: PASS scheduler-runnable-current-keeps-cpu");
2109 }
2110
2111 HostedNewThreadContext reapContext;
2112 const size_t reapBaseline = m_nDeferredThreadReapCompletions.value();
2113 Thread* reapTarget = new Thread(kernelProcess, hostedNewThreadWorkerEntry, &reapContext, nullptr,
2114 false, true, true);
2115 reapTarget->setName("hosted deferred Thread reap target");
2116 const bool reapStarted = reapTarget->startDetached();
2117 bool reapCompleted = false;
2118 for (size_t attempt = 0; attempt < Attempts; ++attempt) {
2119 if (m_nDeferredThreadReapCompletions.value() == reapBaseline + 1) {
2120 reapCompleted = true;
2121 break;
2122 }
2124 }
2125 const bool reapPassed = check(
2126 reapStarted && reapCompleted && reapContext.calls == 1 && !m_nDeferredThreadReaps.value(),
2127 "detached Thread was not destroyed by the ordinary maintenance worker");
2128 passed &= reapPassed;
2129 if (reapPassed) {
2130 NOTICE("HOSTED-WAIT-TEST: PASS perprocessor-deferred-thread-reap");
2131 }
2132
2133 HostedNewThreadContext delayedContext;
2134 Thread* delayed = new Thread(kernelProcess, hostedNewThreadWorkerEntry, &delayedContext, nullptr,
2135 false, true, true);
2136 delayed->setName("hosted delayed add-worker target");
2137 const bool delayedParked = waitUntilParked(delayed);
2138 for (size_t attempt = 0; attempt < 64; ++attempt) {
2140 }
2141 const bool stayedDormant = delayedParked && isParked(delayed) && delayedContext.calls == 0;
2142 const bool started = delayed->start();
2143 const bool delayedJoined = delayed->joinForCompletion();
2144 const bool delayedPassed =
2145 check(stayedDormant && started && delayedJoined && delayedContext.calls == 1,
2146 "delayed add-worker target did not remain parked until its single "
2147 "start publication");
2148 passed &= delayedPassed;
2149 if (delayedPassed) {
2150 NOTICE("HOSTED-WAIT-TEST: PASS perprocessor-delayed-start-wake");
2151 }
2152
2153 HostedNewThreadContext terminatedContext;
2154 Thread* terminated = new Thread(kernelProcess, hostedNewThreadWorkerEntry, &terminatedContext,
2155 nullptr, false, true, true);
2156 terminated->setName("hosted terminated add-worker target");
2157 const bool terminatedParked = waitUntilParked(terminated);
2159 const bool terminatedJoined = terminated->joinForCompletion();
2160 const bool terminatedPassed =
2161 check(terminatedParked && terminatedJoined && terminatedContext.calls == 0,
2162 "terminate-before-start did not retire the parked add-worker target");
2163 passed &= terminatedPassed;
2164 if (terminatedPassed) {
2165 NOTICE("HOSTED-WAIT-TEST: PASS perprocessor-terminate-before-start");
2166 }
2167
2168 HostedNewThreadContext teardownContext;
2169 Thread* teardown = new Thread(kernelProcess, hostedNewThreadWorkerEntry, &teardownContext,
2170 nullptr, false, true, true);
2171 teardown->setName("hosted add-worker teardown target");
2172 const bool teardownParked = waitUntilParked(teardown);
2173
2174 HostedNewThreadContext detachedTeardownContext;
2175 const size_t detachedReapBaseline = m_nDeferredThreadReapCompletions.value();
2176 Thread* detachedTeardown = new Thread(kernelProcess, hostedNewThreadWorkerEntry,
2177 &detachedTeardownContext, nullptr, false, true, true);
2178 detachedTeardown->setName("hosted detached add-worker teardown target");
2179 const bool detachedTeardownParked = waitUntilParked(detachedTeardown);
2180 const bool detachedTeardownClaimed = detachedTeardown->detach();
2181
2182 stopNewThreadWorker();
2183 const bool teardownJoined = teardown->joinForCompletion();
2184 bool detachedTeardownReaped = false;
2185 for (size_t attempt = 0; attempt < Attempts; ++attempt) {
2186 if (m_nDeferredThreadReapCompletions.value() == detachedReapBaseline + 1) {
2187 detachedTeardownReaped = true;
2188 break;
2189 }
2191 }
2192
2193 m_NewThreadDataLock.acquire();
2194 const bool teardownDrained = !m_NewThreadAdmissionOpen && m_StopNewThreadWorker &&
2195 !m_NewThreadData.count() && !m_DelayedNewThreadData.count();
2196 m_NewThreadDataLock.release();
2197 const bool workerJoined = !m_NewThreadWorker;
2198
2199 // A joined worker cannot be hiding on the condition variable or retain a
2200 // detached reference to this scheduler's queue state.
2201 m_NewThreadDataCondition.broadcast();
2202 for (size_t attempt = 0; attempt < 64; ++attempt) {
2204 }
2205
2206 const bool teardownPassed = check(
2207 teardownParked && teardownJoined && teardownContext.calls == 0 && detachedTeardownParked &&
2208 detachedTeardownClaimed && detachedTeardownReaped && detachedTeardownContext.calls == 0 &&
2209 !m_nDeferredThreadReaps.value() && teardownDrained && workerJoined,
2210 "owned add worker did not drain and join with pending parked work");
2211 passed &= teardownPassed;
2212 if (teardownPassed) {
2213 NOTICE("HOSTED-WAIT-TEST: PASS perprocessor-worker-teardown");
2214 }
2215
2216 startNewThreadWorker(kernelProcess);
2217
2218 HostedNewThreadContext restartContext;
2219 Thread* restart = new Thread(kernelProcess, hostedNewThreadWorkerEntry, &restartContext, nullptr,
2220 false, true, true);
2221 restart->setName("hosted restarted add-worker target");
2222 const bool restartParked = waitUntilParked(restart);
2223 const bool restartStarted = restart->start();
2224 bool restartReapable = false;
2225 for (size_t attempt = 0; attempt < Attempts; ++attempt) {
2226 if (restart->isReapableForHostedTest()) {
2227 restartReapable = true;
2228 break;
2229 }
2231 }
2232 const bool restartJoined = restartReapable && restart->joinForCompletion();
2233 const bool restartPassed =
2234 check(restartParked && restartStarted && restartJoined && restartContext.calls == 1,
2235 "replacement add worker did not process a fresh delayed admission");
2236 passed &= restartPassed;
2237 if (restartPassed) {
2238 NOTICE("HOSTED-WAIT-TEST: PASS perprocessor-worker-restart");
2239 }
2240
2241 return passed;
2242}
2243#endif
2244
2246 __atomic_store_n(&m_IdleWakeRequested, true, __ATOMIC_RELEASE);
2247 prompt();
2248}
2249
2250void PerProcessorScheduler::setIdle(Thread* pThread) {
2251 __atomic_store_n(&m_pIdleThread, pThread, __ATOMIC_RELEASE);
2252 if (!pThread)
2253 __atomic_store_n(&m_IdleWakeRequested, false, __ATOMIC_RELEASE);
2254}
2255
2259 updateOneShotTimer();
2260 if (__atomic_load_n(&m_IdleWakeRequested, __ATOMIC_ACQUIRE)) {
2262 return;
2263 }
2265 schedule(Thread::Ready, false);
2266 } else {
2267 // The ready check and STI/HLT are one interrupt-masked handshake. A
2268 // remote IPI arriving after the check remains pending until STI/HLT.
2270 }
2272}
void waitForCompletion(Mutex &mutex)
void reset()
Definition Event.cc:159
Definition Event.h:49
static uintptr_t getTrampoline()
Definition Event.cc:201
virtual bool prefersAlternateUserStack() const
Definition Event.h:249
virtual bool isDeletable()
Definition Event.cc:225
uintptr_t getHandlerAddress()
Definition Event.h:231
HandlerPrivilege getHandlerPrivilege() const
Definition Event.h:236
static size_t getHandlerBufferSize()
Definition Event.cc:217
bool isValidHandlerMapping(size_t mappingFlags) const
Definition Event.cc:229
virtual bool isSignalEvent() const
Definition Event.h:244
virtual UserReturnDelivery deliverAtUserReturn(InterruptState &)
Definition Event.h:259
virtual bool requiresExactUserReturnState() const
Definition Event.h:254
UserReturnDelivery
Definition Event.h:58
static uintptr_t getHandlerBuffer()
Definition Event.cc:213
virtual size_t getNumber()=0
virtual size_t serialize(uint8_t *pBuffer)=0
void push(Node &node)
MUST_USE_RESULT PopResult pop(Node *&out)
Iterator begin()
Definition List.h:122
::Iterator< T, node_t > Iterator
Definition List.h:67
Iterator end()
Definition List.h:132
bool checkSchedule(size_t nCpu=~0U)
bool lockReleased(const Spinlock *pLock, size_t nCpu=~0U)
virtual SchedulerTimer * getSchedulerTimer()=0
virtual Timer * getTimer()=0
MUST_USE_RESULT bool tryEnter()
void join()
Definition OwnedThread.h:49
void setClockDeadline(uint64_t deadline)
void serviceDeferredSubsystemException(InterruptState &state, bool diagnosticSample=false)
void commitCurrentThreadExit(Spinlock *pLock=0) NORETURN
void timer(uint64_t delta, InterruptState &state)
void schedule(Thread::Status nextStatus=Thread::Ready, bool dispatchEvents=true)
SchedulingAlgorithm * m_pSchedulingAlgorithm
void checkEventState(uintptr_t userStack)
void abandonCurrentThreadStack(StackDiscardReason reason, Spinlock *pLock=0) NORETURN
void addThread(Thread *pThread, Thread::ThreadStartFunc pStartFunction, void *pParam, bool bUsermode, void *pStack)
void initialise(Thread *pThread)
MUST_USE_RESULT bool serviceProcessStopAtUserReturn(ProcessStopGateMode mode=ProcessStopGateMode::StopOnly)
void registerWorkerWake(SchedulerWorkerWake &worker, WaitQueue &waiters)
MUST_USE_RESULT bool serviceUserReturnWork(InterruptState &state, UserReturnFrame::Origin origin=UserReturnFrame::Origin::Interrupt, bool diagnosticSample=false)
void unregisterWorkerWake(SchedulerWorkerWake &worker)
void killCurrentThread(Spinlock *pLock=0) NORETURN
static void deleteThreadThenRestoreState(Thread *pThread, SchedulerState &newState, volatile uintptr_t *pLock=0) NORETURN
void publishReadyFromWait(Thread *pThread)
virtual physical_uintptr_t allocatePage(size_t pageConstraints=0)=0
static PhysicalMemoryManager & instance()
ProcessState
Definition Process.h:303
@ Reaped
Terminal wait status is visible; the owner may still be on-stack.
Definition Process.h:308
Process * getParent()
Definition Process.h:620
void publishTerminationReapable()
Definition Process.cc:2385
VirtualAddressSpace * getAddressSpace()
Definition Process.h:530
OperationBarrier m_DeferredThreadReaps
Definition Process.h:1323
Spinlock m_Lock
Definition Process.h:1204
bool terminatingThreadReapable(Thread *pThread, bool &wakeOwner)
Definition Process.cc:2308
WaitQueue m_TerminationWaiters
Definition Process.h:1097
static void restoreState(SchedulerState &state, volatile uintptr_t *pLock=0) NORETURN
static void jumpUser(volatile uintptr_t *pLock, uintptr_t address, uintptr_t stack, uintptr_t p1=0, uintptr_t p2=0, uintptr_t p3=0, uintptr_t p4=0) NORETURN
static bool getInterrupts()
static ProcessorId id()
static void setTlsBase(uintptr_t newBase)
static ProcessorInformation & information()
static bool saveState(SchedulerState &state)
static size_t getCount()
static bool inDeviceHardIrq()
Definition Processor.h:581
static void switchAddressSpace(VirtualAddressSpace &AddressSpace)
static bool guardDeviceHardIrqOperation(DeviceHardIrqOperation operation)
Definition Processor.h:585
static ExecutionContext executionContext()
Definition Processor.cc:109
static void switchState(bool bInterrupts, SchedulerState &a, SchedulerState &b, volatile uintptr_t *pLock=0)
static void jumpKernel(volatile uintptr_t *pLock, uintptr_t address, uintptr_t stack, uintptr_t p1=0, uintptr_t p2=0, uintptr_t p3=0, uintptr_t p4=0) NORETURN
static ProcessorInformation * informationAt(size_t cpu)
Definition Processor.cc:39
static void saveAndJumpKernel(bool bInterrupts, SchedulerState &s, volatile uintptr_t *pLock, uintptr_t address, uintptr_t stack, uintptr_t p1=0, uintptr_t p2=0, uintptr_t p3=0, uintptr_t p4=0)
static void setInterrupts(bool bEnable)
static size_t index()
static void saveAndJumpUser(bool bInterrupts, SchedulerState &s, volatile uintptr_t *pLock, uintptr_t address, uintptr_t stack, uintptr_t p1=0, uintptr_t p2=0, uintptr_t p3=0, uintptr_t p4=0)
static void haltUntilInterrupt()
virtual bool registerHandler(SchedulerTimerHandler *handler)=0
virtual bool removeHandler(SchedulerTimerHandler *handler)=0
void drainDeferredTimeAccounting()
Definition Scheduler.cc:190
static Scheduler & instance()
Definition Scheduler.h:96
void requestLoadAverageSample()
void yield()
Definition Scheduler.cc:236
virtual void removeThread(Thread *pThread)=0
virtual void addThread(Thread *pThread)=0
virtual bool hasReady()=0
virtual Thread * getNext(Thread *pCurrentThread)=0
void release(size_t n=1)
Definition Semaphore.cc:549
bool acquire(size_t n=1, size_t timeoutSecs=0, size_t timeoutUsecs=0)
Definition Semaphore.cc:355
volatile processor_register_t * deferredReleaseWord()
Definition Spinlock.cc:190
void release()
Definition Spinlock.cc:168
bool acquire(bool recurse=false, bool safe=true)
Definition Spinlock.cc:36
void exit(uintptr_t ra=0)
Definition Spinlock.cc:164
virtual bool resolveUserPageFault(Thread &, InterruptState &, uintptr_t, uintptr_t)
Definition Subsystem.h:146
virtual void threadException(Thread *pThread, ExceptionType eType, InterruptState *pState=nullptr, uintptr_t faultAddress=0, uintptr_t errorCode=0)
Definition Subsystem.cc:33
virtual void exit(int code, ExitCause cause=ExitCause::Normal)=0
void recordTime(CpuTimeMode mode)
Definition Thread.cc:384
void unlinkWaitsForStackDiscard()
Definition Thread.cc:640
WaitQueue m_EventWaiters
Definition Thread.h:1288
void accountTimerTick(Time::Timestamp delta, bool kernelMode)
Definition Thread.cc:422
void setUnwindState(UnwindType ut)
Definition Thread.cc:3580
CpuTimeMode currentTimeAccountingMode() const
Definition Thread.cc:471
AlternateSignalStack m_AlternateSignalStack
Definition Thread.h:1302
UnwindType
Definition Thread.h:516
@ Continue
No unwind necessary, carry on as normal.
Definition Thread.h:517
@ TerminateThread
Exit only this thread during Process exit.
Definition Thread.h:519
@ Exit
Exit the owning process at the next safe boundary.
Definition Thread.h:518
SchedulerState & state()
Definition Thread.cc:832
bool interruptWaitUnlocked(WaitQueue::WakeReason reason, PerProcessorScheduler *&readyScheduler)
Definition Thread.cc:3657
DeferredScopeRecord * m_pDeferredScopes[MAX_NESTED_EVENTS]
Definition Thread.h:1390
volatile Status m_Status
Definition Thread.h:1310
void shutdown()
Definition Thread.cc:570
bool m_bStartRequested
Definition Thread.h:1352
size_t m_EventDeferralDepth
Definition Thread.h:1379
bool joinForCompletion()
Definition Thread.cc:2750
MUST_USE_RESULT Event::Delivery getNextEvent(EventSelection selection=EventSelection::AnyDeliverable)
Definition Thread.cc:2466
int(* ThreadStartFunc)(void *)
Definition Thread.h:189
bool markReapable()
Definition Thread.cc:3721
void * getKernelStack()
Definition Thread.cc:1070
Status getStatus() const
Definition Thread.h:433
bool eventsDeferred() const
Definition Thread.cc:3159
UnwindType getUnwindState()
Definition Thread.h:535
bool detach()
Definition Thread.cc:2987
DeferredThreadReapNode m_DeferredReapNode
Definition Thread.h:1189
Spinlock m_Lock
Definition Thread.h:1262
void trackTime(CpuTimeMode mode)
Definition Thread.cc:392
void setScheduler(class PerProcessorScheduler *pScheduler)
Definition Thread.cc:3531
bool startDetached()
Definition Thread.cc:769
Process * getParent() const
Definition Thread.h:340
size_t getId()
Definition Thread.h:465
void closeExternalLeaseAdmission()
Definition Thread.cc:2946
void popState(bool clean=true)
Definition Thread.cc:913
size_t m_nStateLevel
Definition Thread.h:1177
void setCpuId(size_t id)
Definition Thread.h:870
class PerProcessorScheduler * getScheduler() const
Definition Thread.h:929
void setStatus(Status s)
Definition Thread.cc:719
DeferredProcessExit takeDeferredProcessExit()
Definition Thread.cc:3618
Spinlock & getLock()
Definition Thread.h:613
bool isTerminationDeferred() const
Definition Thread.h:569
SchedulerState * pushState()
Definition Thread.cc:841
bool start()
Definition Thread.cc:751
bool sendEvent(Event *pEvent)
Definition Thread.cc:1115
uintptr_t getTlsBase()
Definition Thread.cc:2644
size_t getStateLevel() const
Definition Thread.h:316
size_t m_TerminationDeferralDepth
Definition Thread.h:1387
bool takeDeferredSubsystemException(size_t &type, uintptr_t &faultAddress, uintptr_t &errorCode)
Definition Thread.cc:3643
void markDeferredUserReturnSignalInterruption()
Definition Thread.cc:2551
void abandonCurrentState(bool clean=false)
Definition Thread.cc:956
void transitionTime(CpuTimeMode from, CpuTimeMode to, bool interruptsAlreadyDisabled=false)
Definition Thread.cc:405
bool m_ExitToIdle
Definition Thread.h:1314
virtual Stack * allocateStack()=0
virtual bool map(physical_uintptr_t physicalAddress, void *virtualAddress, size_t flags)=0
virtual bool isMapped(void *virtualAddress)=0
virtual uintptr_t getUserStart() const =0
virtual bool getMapping(void *virtualAddress, physical_uintptr_t &physicalAddress, size_t &flags)=0
virtual uintptr_t getKernelStart() const =0
virtual bool isAddressValid(void *virtualAddress)=0
void EXPORTED_PUBLIC panic(const char *msg) NORETURN
Definition panic.cc:118
@ Hex
Definition Log.h:124
Iterator erase(Iterator &Iter)
Definition List.h:352
T popFront()
Definition List.h:330
size_t count() const
Definition List.h:212
void pushBack(const T &value)
Definition List.h:216
bool m_bOwnsAlternateSignalStack
Definition Thread.h:1142