The Pedigree Project 0.1
x64/SyscallManager.cc
1/*
2 * Copyright (c) 2008-2014, Pedigree Developers
3 *
4 * Please see the CONTRIB file in the root of the source tree for a full
5 * list of contributors.
6 *
7 * Permission to use, copy, modify, and distribute this software for any
8 * purpose with or without fee is hereby granted, provided that the above
9 * copyright notice and this permission notice appear in all copies.
10 *
11 * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
12 * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
13 * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
14 * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
15 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
16 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
17 * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
18 */
19
20#include "SyscallManager.h"
21#include "pedigree/kernel/ActivityDiagnostics.h"
22#include "pedigree/kernel/Log.h"
23#include "pedigree/kernel/Metrics.h"
24#include "pedigree/kernel/Subsystem.h"
25#include "pedigree/kernel/compiler.h"
26#include "pedigree/kernel/process/PerProcessorScheduler.h"
27#include "pedigree/kernel/process/Process.h"
28#include "pedigree/kernel/process/TerminationDeferral.h"
29#include "pedigree/kernel/process/Thread.h"
30#include "pedigree/kernel/process/TimeTracker.h"
31#include "pedigree/kernel/processor/Processor.h"
32#include "pedigree/kernel/processor/ProcessorInformation.h"
33#include "pedigree/kernel/processor/SyscallHandler.h"
34#include "pedigree/kernel/processor/state.h"
35#include "pedigree/kernel/syscallError.h"
36
38
39extern void system_reboot(Machine::ShutdownType type);
40extern "C" void pedigree_defer_user_entry(X64UserEntryMetadata*);
41
42namespace {
43void captureUserEntry(SyscallState& state) {
44 pedigree_defer_user_entry(&state.m_UserEntry);
45}
46
47class SyscallReturnScope {
48 public:
49 explicit SyscallReturnScope(const SyscallState* state)
50 : m_Thread(Processor::information().getCurrentThread()),
51 m_StateLevel(m_Thread->getStateLevel()),
52 m_Previous(m_Thread->getOriginalSyscallState()),
53 m_Discard(&restore, this) {
54 m_Thread->setOriginalSyscallState(state);
55 }
56
57 ~SyscallReturnScope() {
58 restore(this);
59 }
60
61 private:
62 static void restore(void* context) {
63 SyscallReturnScope* scope = static_cast<SyscallReturnScope*>(context);
64 if (scope->m_Thread) {
65 scope->m_Thread->restoreDeferredSignalMask(scope->m_StateLevel);
66 scope->m_Thread->setOriginalSyscallState(scope->m_Previous);
67 scope->m_Thread = nullptr;
68 }
69 }
70
71 Thread* m_Thread;
72 size_t m_StateLevel;
73 const SyscallState* m_Previous;
75};
76
77void recordAffinitySample(bool active, bool waited, uint64_t start) {
78 if (!active)
79 return;
80 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::SyscallAffinity,
81 ActivityDiagnostics::timestamp() - start);
82 if (waited)
83 ActivityDiagnostics::recordUserReturnAffinityWait(true);
84}
85
86void recordAccountingSample(bool active, uint64_t start) {
87 if (active) {
88 ActivityDiagnostics::recordUserReturnStage(
89 ActivityDiagnostics::UserReturnStage::SyscallAccounting,
90 ActivityDiagnostics::timestamp() - start);
91 }
92}
93
94bool finishAffinityReturn(SyscallState& state, const SyscallState* original,
95 bool diagnosticSample) {
96 Thread* current = Processor::information().getCurrentThread();
97 while (true) {
98 bool waited = false;
99 const uint64_t affinityStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
100 const bool terminal = current->completeAffinityAtSafePoint(&waited) == AffinityResult::Terminal;
101 recordAffinitySample(diagnosticSample, waited, affinityStart);
102 if (terminal)
103 return true;
104 if (!waited)
105 return false;
107 SyscallReturnScope returnScope(original);
108 if (Processor::information().getScheduler().serviceUserReturnWork(
109 state, UserReturnFrame::Origin::Syscall, diagnosticSample))
110 return true;
111 }
112}
113
114bool finishAffinityReturn(InterruptState& state, UserReturnFrame::Origin origin,
115 bool diagnosticSample) {
116 Thread* current = Processor::information().getCurrentThread();
117 while (true) {
118 bool waited = false;
119 const uint64_t affinityStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
120 const bool terminal = current->completeAffinityAtSafePoint(&waited) == AffinityResult::Terminal;
121 recordAffinitySample(diagnosticSample, waited, affinityStart);
122 if (terminal)
123 return true;
124 if (!waited)
125 return false;
127 if (Processor::information().getScheduler().serviceUserReturnWork(state, origin,
128 diagnosticSample))
129 return true;
130 }
131}
132} // namespace
133
136}
137
139 Registration& registration, FastEntry entry) {
140 return registerHandler(Service, pHandler, registration, entry);
141}
142
143void X64SyscallManager::syscall(SyscallState& syscallState) {
144 Metrics::increment(Metrics::Syscall);
145 captureUserEntry(syscallState);
146 // Restart handling consumes only the entry registers, not deferred FS/GS bases.
147 const SyscallState originalState = syscallState;
148#if PEDIGREE_ACTIVITY_DIAGNOSTICS
149 const bool diagnosticSample =
150 Processor::information().getScheduler().sampleUserReturnDiagnostics();
151#else
152 const bool diagnosticSample = false;
153#endif
154 Thread* syscallThread = Processor::information().getCurrentThread();
155 bool commitThreadExit = false;
156 bool exitCurrentProcess = false;
157 bool rebootSystem = false;
158 Machine::ShutdownType shutdownType = Machine::ShutdownType::Halt;
159 bool userReturnTerminal = false;
160 bool interruptedWithoutProgress = false;
161 bool deferTimeAccountingToUserReturn = false;
162 uint64_t returnTailStart = 0;
163 int processExitCode = 0;
164 Subsystem::ExitCause processExitCause = Subsystem::ExitCause::Normal;
165 {
166 // SYSCALL entered with IF masked by IA32_FMASK. Let the first accounting
167 // sample reuse that architectural state instead of masking and restoring
168 // interrupts a second time.
169 TimeTracker tracker(0, true, true, syscallThread);
170
171 // Enable IRQs - stack switching and such are done now and it's now safe to
172 // start processing interrupts elsewhere.
174
175 size_t serviceNumber = syscallState.getSyscallService();
176#if PEDIGREE_BENCHMARK_SYSCALL_TIMING
177 if (serviceNumber == linuxCompat) {
178 tracker.attributeSyscall(syscallState.getSyscallNumber());
179 }
180#endif
181 bool handled = false;
182 PostSyscallAction action;
183 if (LIKELY(serviceNumber < serviceEnd)) {
184 // Blocking callbacks must finish ownership waits before a terminal
185 // request can consume this thread's stack.
186 HandlerLease handler;
187 if (m_Instance.acquireHandler(static_cast<Service_t>(serviceNumber), handler, action)) {
188 handled = true;
189 uint64_t result = m_Instance.dispatchHandler(handler, syscallState);
190 uint64_t errno = syscallThread->getErrno();
191 interruptedWithoutProgress = result == static_cast<uint64_t>(-1) &&
192 errno == Error::Interrupted && serviceNumber == linuxCompat;
195 if (serviceNumber == linuxCompat) {
196 if (errno != 0) {
197 syscallState.setSyscallReturnValue(-errno);
198 } else {
199 syscallState.setSyscallReturnValue(result);
200 }
201 } else {
202 syscallState.setSyscallReturnValue(result);
203 syscallState.setSyscallErrno(errno);
204 }
205 // Reset error number now that we've extracted it.
206 syscallThread->setErrno(0);
207 }
208 }
209
210 returnTailStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
211 if (!handled) {
212 // Even an invalid service or a temporarily missing handler came from a
213 // real userspace frame and must not bypass pending return work.
214 userReturnTerminal = Processor::information().getScheduler().serviceUserReturnWork(
215 syscallState, UserReturnFrame::Origin::Syscall, diagnosticSample);
216 } else {
217 const bool directUserTransition =
218 action.kind == ReturnFromEvent || action.kind == PopEventState ||
219 action.kind == RestoreProcessorState || action.kind == JumpToUserspace;
220 if (directUserTransition) {
221 const uint64_t stopStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
222 userReturnTerminal = Processor::information().getScheduler().serviceProcessStopAtUserReturn(
223 PerProcessorScheduler::ProcessStopGateMode::DirectUserTransition);
224 if (diagnosticSample) {
225 ActivityDiagnostics::recordUserReturnStage(
226 ActivityDiagnostics::UserReturnStage::ProcessStop,
227 ActivityDiagnostics::timestamp() - stopStart);
228 }
229 Thread* current = syscallThread;
230 if (current && current->getUnwindState() != Thread::Continue) {
231 userReturnTerminal = true;
232 }
233 }
234
235 switch (action.kind) {
236 case TerminateCurrentThread:
237 commitThreadExit = true;
238 break;
239 case ExitCurrentProcess:
240 exitCurrentProcess = true;
241 processExitCode = static_cast<int>(action.value);
242 break;
243 case ReturnFromEvent:
244 if (userReturnTerminal) {
245 break;
246 }
247 tracker.finishInKernel();
248 Processor::information().getScheduler().eventHandlerReturned();
249 break;
250 case PopEventState:
251 if (!userReturnTerminal) {
252 syscallThread->abandonCurrentState(false);
253 }
254 break;
255 case RestoreProcessorState: {
256 if (userReturnTerminal) {
257 break;
258 }
259 // Linux rt_sigreturn replaces the current user register
260 // image; it does not own a Pedigree event state to pop.
261 const uintptr_t userStack = action.state.rsp;
262 const uint64_t userFlags = action.state.rflags;
263 alignas(16) unsigned char interruptStack[sizeof(InterruptState)] = {};
264 action.state.setStackPointer(
265 reinterpret_cast<uintptr_t>(interruptStack + sizeof(interruptStack)));
266 X64UserEntryMetadata metadata = syscallState.getUserEntryMetadata();
267 metadata.origRax = ~uint64_t(0);
268 InterruptState* returnState = InterruptState::construct(action.state, true, metadata);
269 returnState->setStackPointer(userStack);
270 returnState->setFlags(userFlags);
271 // rt_sigreturn restores the old mask before committing this frame.
272 // Service newly unblocked signals against that exact restored image
273 // so none escape briefly to userspace or wait for another syscall.
274 userReturnTerminal = Processor::information().getScheduler().serviceUserReturnWork(
275 *returnState, UserReturnFrame::Origin::SignalRestore, diagnosticSample);
276 if (userReturnTerminal) {
277 break;
278 }
279 tracker.finishInKernel();
280 Thread* current = syscallThread;
281 if (finishAffinityReturn(*returnState, UserReturnFrame::Origin::SignalRestore,
282 diagnosticSample)) {
283 userReturnTerminal = true;
285 break;
286 }
287 const uint64_t accountingStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
288 current->transitionTime(CpuTimeMode::Kernel, CpuTimeMode::User);
289 recordAccountingSample(diagnosticSample, accountingStart);
290 Processor::contextSwitch(returnState);
291 }
292 case JumpToUserspace: {
293 if (userReturnTerminal)
294 break;
295 tracker.finishInKernel();
297 Thread* current = syscallThread;
298 current->abandonAllStates();
299 SyscallState newImage;
300 ByteSet(&newImage, 0, sizeof(newImage));
301 newImage.setInstructionPointer(action.state.getInstructionPointer());
302 newImage.setStackPointer(action.state.getStackPointer());
303 newImage.setFlags(0x202);
304 X64UserEntryMetadata metadata = {};
305 metadata.ds = metadata.es = 0x23;
306 metadata.origRax = 59;
307 newImage.setUserEntryMetadata(metadata);
308 // invoke reset TLS to the new image's actual scheduler-owned base.
309 newImage.refreshUserTlsBase();
311 // Materialize the loader-owned stack before the final IRQ-off tail.
312 *reinterpret_cast<volatile uint64_t*>(newImage.getStackPointer() - 8) = 0;
313 while (true) {
314 userReturnTerminal = Processor::information().getScheduler().serviceUserReturnWork(
315 newImage, UserReturnFrame::Origin::NewImage, diagnosticSample);
316 if (userReturnTerminal)
317 break;
318 bool waited = false;
319 const uint64_t affinityStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
320 userReturnTerminal =
321 current->completeAffinityAtSafePoint(&waited) == AffinityResult::Terminal;
322 recordAffinitySample(diagnosticSample, waited, affinityStart);
323 if (userReturnTerminal || !waited)
324 break;
326 }
327 if (userReturnTerminal) {
329 break;
330 }
331 const uint64_t accountingStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
332 current->transitionTime(CpuTimeMode::Kernel, CpuTimeMode::User);
333 recordAccountingSample(diagnosticSample, accountingStart);
334 // This tail restores the same image exposed by the stop and retains
335 // jumpUser's CR0.TS lazy-FPU setup.
336 Processor::restoreState(newImage, nullptr);
337 }
338 case RebootSystem:
339 rebootSystem = true;
340 shutdownType = static_cast<Machine::ShutdownType>(action.value);
341 break;
342 case NoPostSyscallAction: {
343#if PEDIGREE_FAST_USER_RETURN
344 Thread* current = syscallThread;
345 if (!interruptedWithoutProgress && current && current->canSkipUserReturnWork()) {
346 deferTimeAccountingToUserReturn = true;
347 break;
348 }
349#endif
350 SyscallReturnScope returnScope(interruptedWithoutProgress ? &originalState : nullptr);
352 syscallState, UserReturnFrame::Origin::Syscall, diagnosticSample);
353 break;
354 }
355 }
356 }
357
358 if (!exitCurrentProcess && !rebootSystem) {
359 Thread* pThread = syscallThread;
360 const Thread::UnwindType unwindState = pThread->getUnwindState();
361 if (userReturnTerminal || unwindState != Thread::Continue) {
362 if (unwindState == Thread::TerminateThread) {
363 commitThreadExit = true;
364 }
365 if (unwindState == Thread::Exit) {
366 NOTICE("Unwind state exit at syscall return");
367 exitCurrentProcess = true;
368 const Thread::DeferredProcessExit request = pThread->takeDeferredProcessExit();
369 processExitCode = request.code;
370 processExitCause = request.cause;
371 }
372 }
373 }
374
375 // Make sure we come back out with interrupts enabled at all times.
376 if ((syscallState.m_RFlagsR11 & 0x200) != 0x200) {
377 syscallState.m_RFlagsR11 |= 0x200;
378 }
379
380 if (deferTimeAccountingToUserReturn)
381 tracker.finishForUserReturn();
382 else
383 tracker.finishInKernel();
384 }
385
386 if (rebootSystem) {
388 syscallThread->abandonAllStates();
390 system_reboot(shutdownType);
391 return;
392 }
393 if (exitCurrentProcess) {
394 syscallThread->getParent()->getSubsystem()->exit(processExitCode, processExitCause);
395 }
396 if (commitThreadExit) {
397 Processor::information().getScheduler().commitCurrentThreadExit();
398 }
399
400 // No syscall handler, return-action lease, or accounting scope survives
401 // this boundary. Keep IRQs disabled from the final mask check through SYSRET.
402 Processor::information().getScheduler().servicePendingScheduling();
403 Thread* current = syscallThread;
404 if (finishAffinityReturn(syscallState, interruptedWithoutProgress ? &originalState : nullptr,
405 diagnosticSample)) {
407 Processor::information().getScheduler().commitUserReturnTerminalState();
408 FATAL_NOLOCK("Terminal affinity return unexpectedly returned");
409 }
410 const uint64_t accountingStart = diagnosticSample ? ActivityDiagnostics::timestamp() : 0;
411 // completeAffinityAtSafePoint() leaves the ordinary return tail with IRQs
412 // masked. Do not sample and restore that state again before SYSRET.
413 current->transitionTimeAtInterruptReturn(CpuTimeMode::Kernel, CpuTimeMode::User);
414 recordAccountingSample(diagnosticSample, accountingStart);
415 if (diagnosticSample) {
416 ActivityDiagnostics::recordUserReturnStage(ActivityDiagnostics::UserReturnStage::SyscallTail,
417 ActivityDiagnostics::timestamp() - returnTailStart);
418 }
419}
420
421uintptr_t X64SyscallManager::syscall(Service_t service, uintptr_t function, uintptr_t p1,
422 uintptr_t p2, uintptr_t p3, uintptr_t p4, uintptr_t p5) {
423 uint64_t rax = (static_cast<uint64_t>(service) << 16) | function;
424 uint64_t ret;
425 asm volatile(
426 "mov %6, %%r8; \
427 syscall"
428 : "=a"(ret)
429 : "0"(rax), "b"(p1), "d"(p2), "S"(p3), "D"(p4), "m"(p5)
430 : "rcx", "r11");
431 return ret;
432}
433
434//
435// Functions only usable in the kernel initialisation phase
436//
437
438extern "C" void syscall_handler();
440 // Enable SCE (= System Call Extensions)
441 // Set IA32_EFER/EFER.SCE
443 0xC0000080, Processor::readMachineSpecificRegister(0xC0000080) | 0x0000000000000001);
444
445 // Setup SYSCALL/SYSRET
446 // Set the IA32_STAR/STAR (CS/SS segment selectors)
447 Processor::writeMachineSpecificRegister(0xC0000081, 0x001B000800000000LL);
448 // Set the IA32_LSTAR/LSTAR (RIP)
449 Processor::writeMachineSpecificRegister(0xC0000082, reinterpret_cast<uint64_t>(syscall_handler));
450 // Set the IA32_FMASK/SF_MASK (RFLAGS mask, RFLAGS.IF,TF,DF cleared after
451 // syscall)
452 Processor::writeMachineSpecificRegister(0xC0000084, 0x0000000000000700LL);
453}
454
MUST_USE_RESULT bool serviceUserReturnWork(InterruptState &state, UserReturnFrame::Origin origin=UserReturnFrame::Origin::Interrupt, bool diagnosticSample=false)
static void restoreState(SchedulerState &state, volatile uintptr_t *pLock=0) NORETURN
static ProcessorInformation & information()
static void contextSwitch(InterruptState *state) NORETURN
static void setInterrupts(bool bEnable)
virtual void exit(int code, ExitCause cause=ExitCause::Normal)=0
static EXPORTED_PUBLIC SyscallManager & instance()
void setErrno(size_t err)
Definition Thread.h:482
UnwindType
Definition Thread.h:516
@ Continue
No unwind necessary, carry on as normal.
Definition Thread.h:517
@ TerminateThread
Exit only this thread during Process exit.
Definition Thread.h:519
@ Exit
Exit the owning process at the next safe boundary.
Definition Thread.h:518
size_t getErrno()
Definition Thread.h:477
UnwindType getUnwindState()
Definition Thread.h:535
ALWAYS_INLINE void transitionTimeAtInterruptReturn(CpuTimeMode from, CpuTimeMode to)
Definition Thread.h:373
Process * getParent() const
Definition Thread.h:340
class PerProcessorScheduler * getScheduler() const
Definition Thread.h:929
DeferredProcessExit takeDeferredProcessExit()
Definition Thread.cc:3618
AffinityResult completeAffinityAtSafePoint(bool *waited=nullptr)
void abandonAllStates()
Definition Thread.cc:965
void abandonCurrentState(bool clean=false)
Definition Thread.cc:956
void transitionTime(CpuTimeMode from, CpuTimeMode to, bool interruptsAlreadyDisabled=false)
Definition Thread.cc:405
void attributeSyscall(size_t rawNumber)
void finishInKernel()
ALWAYS_INLINE void finishForUserReturn()
Definition TimeTracker.h:66
uintptr_t syscall(Service_t service, uintptr_t function, uintptr_t p1, uintptr_t p2, uintptr_t p3, uintptr_t p4, uintptr_t p5)
X64SyscallManager() INITIALISATION_ONLY
static void initialiseProcessor() INITIALISATION_ONLY
static X64SyscallManager & instance()
virtual bool registerSyscallHandler(Service_t Service, SyscallHandler *pHandler, Registration &registration, FastEntry entry=nullptr)
static void writeMachineSpecificRegister(uint32_t index, uint64_t value)
static uint64_t readMachineSpecificRegister(uint32_t index)