The Pedigree Project 0.1
net-syscalls.cc
1/*
2 * Copyright (c) 2008-2014, Pedigree Developers
3 *
4 * Please see the CONTRIB file in the root of the source tree for a full
5 * list of contributors.
6 *
7 * Permission to use, copy, modify, and distribute this software for any
8 * purpose with or without fee is hereby granted, provided that the above
9 * copyright notice and this permission notice appear in all copies.
10 *
11 * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
12 * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
13 * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
14 * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
15 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
16 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
17 * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
18 */
19
20#define LWIP_DONT_PROVIDE_BYTEORDER_FUNCTIONS 1 // don't need them here
21
22#include "pedigree/kernel/Subsystem.h"
23#include "pedigree/kernel/process/Process.h"
24#include "pedigree/kernel/process/Scheduler.h"
25#include "pedigree/kernel/process/TerminationDeferral.h"
26#include "pedigree/kernel/process/Thread.h"
27#include "pedigree/kernel/processor/Processor.h"
28#include "pedigree/kernel/processor/types.h"
29#include "pedigree/kernel/syscallError.h"
30#include "pedigree/kernel/utilities/HashTable.h"
31#include "pedigree/kernel/utilities/Pointers.h"
32#include "pedigree/kernel/utilities/Tree.h"
33#include "pedigree/kernel/utilities/UniqueResource.h"
34
35#include <errno.h>
36#include <fcntl.h>
37#include <limits.h>
38#include <stddef.h>
39
40#include "eventfd-syscalls.h"
41#include "file-syscalls.h"
42#include "landlock.h"
43#include "modules/subsys/posix/FileDescriptor.h"
44#include "modules/subsys/posix/PosixSubsystem.h"
45#include "modules/subsys/posix/ResolvedPath.h"
46#include "modules/subsys/posix/UnixFilesystem.h"
52#include "modules/system/vfs/File.h"
53#include "modules/system/vfs/MountView.h"
54#include "modules/system/vfs/VFS.h"
55#include "mqueue-netlink.h"
56#include "net-syscalls.h"
57#include "recvmmsg-syscalls.h"
58#include "sandbox-state.h"
59#include "signalfd-syscalls.h"
60#include "timerfd-syscalls.h"
61
62#ifndef UTILITY_LINUX
63#include <netdb.h>
64
65#include <netinet/tcp.h>
66#include <sys/socket.h>
67#endif
68
69#include <netinet/in.h>
70#include <sys/un.h>
71
72Tree<struct netconn*, LwipSocketSyscalls*> LwipSocketSyscalls::m_SyscallObjects;
73Mutex LwipSocketSyscalls::m_SyscallObjectsLock;
74Tree<UnixSocket*, UnixSocketSyscalls*> UnixSocketSyscalls::m_SyscallObjects;
75Tree<UnixSocket*, UnixSocket*> UnixSocketSyscalls::m_Peers;
76Tree<UnixSocket*, UnixSocket*> UnixSocketSyscalls::m_PendingListeners;
77Mutex UnixSocketSyscalls::m_SyscallObjectsLock;
78
79namespace {
80struct NetbufReleaser {
81 static void release(struct netbuf* buffer) {
82 netbuf_delete(buffer);
83 }
84};
85
87
88class SocketPayload {
89 public:
90 bool prepare(const struct msghdr& message, int type, int domain, bool sending,
91 bool kernelBuffer = false) {
92 size_t requested = 0;
93 for (size_t i = 0; i < static_cast<size_t>(message.msg_iovlen); ++i) {
94 if (message.msg_iov[i].iov_len > static_cast<size_t>(SSIZE_MAX) - requested) {
95 SYSCALL_ERROR(InvalidArgument);
96 return false;
97 }
98 requested += message.msg_iov[i].iov_len;
99 }
100 size_t capacity = requested;
101 // Streams may make a short transfer. Packet writes must remain indivisible.
102 if (type == SOCK_STREAM && capacity > 65536) {
103 capacity = 65536;
104 } else if (domain == AF_UNIX && type == SOCK_SEQPACKET && capacity > MAX_UNIX_STREAM_QUEUE) {
105 if (sending) {
106 syscallError(EMSGSIZE);
107 return false;
108 }
109 capacity = MAX_UNIX_STREAM_QUEUE;
110 } else if (domain == AF_INET && type == SOCK_DGRAM && capacity > 65535) {
111 if (sending) {
112 syscallError(EMSGSIZE);
113 return false;
114 }
115 capacity = 65535;
116 } else if (domain == 16 && !sending && capacity > 65536) {
117 capacity = 65536;
118 }
119 if (capacity) {
120 m_Bytes = UniqueArray<uint8_t>::allocate(capacity);
121 if (!m_Bytes) {
122 SYSCALL_ERROR(OutOfMemory);
123 return false;
124 }
125 }
126 m_Vector = {m_Bytes.get(), capacity};
127 if (!sending) {
128 return true;
129 }
130 size_t copied = 0;
131 for (size_t i = 0; i < static_cast<size_t>(message.msg_iovlen) && copied < capacity; ++i) {
132 const size_t amount = message.msg_iov[i].iov_len < capacity - copied
133 ? message.msg_iov[i].iov_len
134 : capacity - copied;
135 if (kernelBuffer) {
136 MemoryCopy(m_Bytes.get() + copied, message.msg_iov[i].iov_base, amount);
137 } else if (!PosixSubsystem::copyFromUser(m_Bytes.get() + copied, message.msg_iov[i].iov_base,
138 amount)) {
139 SYSCALL_ERROR(BadAddress);
140 return false;
141 }
142 copied += amount;
143 }
144 return true;
145 }
146
147 void attach(struct msghdr& message) {
148 message.msg_iov = &m_Vector;
149 message.msg_iovlen = 1;
150 }
151
152 bool copyReceived(const struct msghdr& target, size_t received) {
153 // MSG_TRUNC can report the packet length beyond the copied payload.
154 const size_t length = received < m_Vector.iov_len ? received : m_Vector.iov_len;
155 size_t copied = 0;
156 for (size_t i = 0; i < static_cast<size_t>(target.msg_iovlen) && copied < length; ++i) {
157 const size_t amount =
158 target.msg_iov[i].iov_len < length - copied ? target.msg_iov[i].iov_len : length - copied;
159 if (!PosixSubsystem::copyToUser(target.msg_iov[i].iov_base, m_Bytes.get() + copied, amount)) {
160 SYSCALL_ERROR(BadAddress);
161 return false;
162 }
163 copied += amount;
164 }
165 return true;
166 }
167
168 private:
169 UniqueArray<uint8_t> m_Bytes;
170 struct iovec m_Vector = {};
171};
172
173bool copySocketAddress(const struct sockaddr_storage* address, socklen_t length,
174 struct sockaddr_storage& result) {
175 if (length < sizeof(sa_family_t) || length > sizeof(result)) {
176 SYSCALL_ERROR(InvalidArgument);
177 return false;
178 }
179 if (!PosixSubsystem::copyFromUser(&result, address, length)) {
180 SYSCALL_ERROR(BadAddress);
181 return false;
182 }
183 if ((result.ss_family == AF_INET && length < sizeof(struct sockaddr_in)) ||
184 (result.ss_family == AF_INET6 && length < sizeof(struct sockaddr_in6))) {
185 SYSCALL_ERROR(InvalidArgument);
186 return false;
187 }
188 return true;
189}
190
191bool validateSocketMessageFlags(int flags, bool sending, int domain = 0, int type = 0) {
192 int supported = sending ? 0 : MSG_DONTWAIT;
193 if (sending && domain == AF_UNIX && type == SOCK_SEQPACKET) {
194 supported |= MSG_DONTWAIT;
195 }
196#ifdef MSG_NOSIGNAL
197 if (sending || domain == 16) {
198 // Socket writes do not currently raise SIGPIPE, so suppression requires
199 // no additional backend action.
200 supported |= MSG_NOSIGNAL;
201 }
202#else
203 (void)sending;
204#endif
205#ifdef MSG_WAITALL
206 if (!sending && domain == 16) {
207 // musl receives its fixed-size SIGEV_THREAD cookie with MSG_WAITALL.
208 supported |= MSG_WAITALL;
209 }
210#endif
211#ifdef MSG_TRUNC
212 if (!sending) {
213 supported |= MSG_TRUNC;
214 }
215#endif
216#ifdef MSG_CMSG_CLOEXEC
217 if (!sending) {
218 supported |= MSG_CMSG_CLOEXEC;
219 }
220#endif
221
222 if (flags & ~supported) {
223 SYSCALL_ERROR(OperationNotSupported);
224 return false;
225 }
226 return true;
227}
228
229constexpr size_t MaximumControlBytes = CMSG_SPACE(SocketRights::MaximumDescriptors * sizeof(int));
230
231bool parseSocketRights(const struct msghdr& message, SharedPointer<SocketRights>& rights) {
232 rights.reset();
233 const size_t controlLength = static_cast<size_t>(message.msg_controllen);
234 if (!controlLength) {
235 return true;
236 }
237 if (!message.msg_control) {
238 SYSCALL_ERROR(BadAddress);
239 return false;
240 }
241 if (controlLength > MaximumControlBytes) {
242 SYSCALL_ERROR(InvalidArgument);
243 return false;
244 }
245
247 if (!PosixSubsystem::copyFromUser(control.get(), message.msg_control, controlLength)) {
248 SYSCALL_ERROR(BadAddress);
249 return false;
250 }
251
252 constexpr size_t HeaderLength = CMSG_LEN(0);
253 if (controlLength < HeaderLength) {
254 SYSCALL_ERROR(InvalidArgument);
255 return false;
256 }
257
258 struct cmsghdr header = {};
259 MemoryCopy(&header, control.get(), sizeof(header));
260 const size_t recordLength = static_cast<size_t>(header.cmsg_len);
261 if (recordLength < HeaderLength || recordLength > controlLength) {
262 SYSCALL_ERROR(InvalidArgument);
263 return false;
264 }
265 if (header.cmsg_level != SOL_SOCKET || header.cmsg_type != SCM_RIGHTS) {
266 SYSCALL_ERROR(OperationNotSupported);
267 return false;
268 }
269
270 const size_t alignedRecordLength = CMSG_ALIGN(recordLength);
271 if ((controlLength != recordLength && controlLength != alignedRecordLength) ||
272 alignedRecordLength < recordLength) {
273 SYSCALL_ERROR(InvalidArgument);
274 return false;
275 }
276
277 const size_t descriptorBytes = recordLength - HeaderLength;
278 if (!descriptorBytes || (descriptorBytes % sizeof(int))) {
279 SYSCALL_ERROR(InvalidArgument);
280 return false;
281 }
282 const size_t descriptorCount = descriptorBytes / sizeof(int);
283 if (descriptorCount > SocketRights::MaximumDescriptors) {
284 SYSCALL_ERROR(InvalidArgument);
285 return false;
286 }
287
288 if (!SocketRights::create(descriptorCount, rights)) {
289 SYSCALL_ERROR(TooManyReferences);
290 return false;
291 }
292
293 PosixSubsystem* subsystem = getSubsystem();
294 if (!subsystem) {
295 rights.reset();
296 SYSCALL_ERROR(BadFileDescriptor);
297 return false;
298 }
299
300 const uint8_t* descriptorData = control.get() + HeaderLength;
301 for (size_t i = 0; i < descriptorCount; ++i) {
302 int fd = -1;
303 MemoryCopy(&fd, descriptorData + (i * sizeof(fd)), sizeof(fd));
304 DescriptorLease descriptor;
305 if (fd < 0 || !subsystem->acquireFileDescriptor(static_cast<size_t>(fd), descriptor)) {
306 rights.reset();
307 SYSCALL_ERROR(BadFileDescriptor);
308 return false;
309 }
310 if (descriptor->epollImpl ||
311 (descriptor->networkImpl && descriptor->networkImpl->getDomain() == AF_UNIX)) {
312 rights.reset();
313 SYSCALL_ERROR(OperationNotSupported);
314 return false;
315 }
316
317 FileDescriptor* transferred = new FileDescriptor(*descriptor);
318 if ((descriptor->networkImpl && !transferred->networkPublished()) ||
319 (descriptor->getEventFdImpl() && !transferred->eventFdPublished()) ||
320 (descriptor->getTimerFdImpl() && !transferred->timerFdPublished()) ||
321 (descriptor->getSignalFdImpl() && !transferred->signalFdPublished())) {
322 delete transferred;
323 rights.reset();
324 SYSCALL_ERROR(BadFileDescriptor);
325 return false;
326 }
327 transferred->fd = ~static_cast<size_t>(0);
328 transferred->setFlags(0);
329 rights->append(transferred);
330 }
331
332 return true;
333}
334} // namespace
335
336static File* findTrackedUnixSocket(const String& pathname) {
337 ResolvedPath fileLease;
338 File* file = findFilePath(pathname, fileLease, FilesystemPathRef(), true);
339 if (file && !file->retainVfsReference()) {
340 return nullptr;
341 }
342 return file;
343}
344
345static void releaseTrackedUnixSocket(File* file) {
346 if (!file) {
347 return;
348 }
349
350 file->releaseVfsReference();
351}
352
353static Thread* beginInterruptibleSocketCall() {
354#if defined(PEDIGREE_EXTERNAL_SOURCE)
355 // The standalone syscall harness has no Pedigree Thread or event source.
356 return nullptr;
357#else
358 Thread* thread = Processor::information().getCurrentThread();
359 thread->clearInterruption();
360 return thread;
361#endif
362}
363
364bool finishInterruptibleSocketCall(Thread* thread, ssize_t result) {
365#if defined(PEDIGREE_EXTERNAL_SOURCE)
366 (void)thread;
367 (void)result;
368 return true;
369#else
370 const bool interrupted = thread->getInterruptionReason() == Thread::InterruptedBySignal;
371 thread->clearInterruption();
372
373 if (interrupted && result < 0) {
374 SYSCALL_ERROR(Interrupted);
375 return false;
376 }
377
378 return true;
379#endif
380}
381
382static bool isSaneSocket(const DescriptorLease& f) {
383 if (!f) {
384 N_NOTICE(" -> isSaneSocket: descriptor is null");
385 SYSCALL_ERROR(BadFileDescriptor);
386 return false;
387 }
388
389 if (!f->networkImpl) {
390 N_NOTICE(" -> isSaneSocket: no network implementation found");
391 syscallError(ENOTSOCK);
392 return false;
393 }
394
395 return true;
396}
397
398static const int socketTypeMask = 0xF;
399static const int socketCreationFlags = SOCK_NONBLOCK | SOCK_CLOEXEC;
400static const int linuxTcpNoDelay = 1;
401
402static bool splitSocketType(int argument, int& type, int& flags) {
403 type = argument & socketTypeMask;
404 flags = argument & socketCreationFlags;
405 if (argument != (type | flags)) {
406 SYSCALL_ERROR(InvalidArgument);
407 return false;
408 }
409
410 return true;
411}
412
413static void setSocketDescriptorFlags(FileDescriptor* descriptor, int flags) {
414 descriptor->setFlags((flags & SOCK_CLOEXEC) ? FD_CLOEXEC : 0);
415 descriptor->setStatusFlags((flags & SOCK_NONBLOCK) ? O_NONBLOCK : 0);
416}
417
418static bool unixSocketPath(const struct sockaddr_storage* address, socklen_t addressLength,
419 String& path, bool allowUnnamed) {
420 const size_t pathOffset = offsetof(struct sockaddr_un, sun_path);
421 if (!address || addressLength < pathOffset || addressLength > sizeof(struct sockaddr_un)) {
422 SYSCALL_ERROR(InvalidArgument);
423 return false;
424 }
425
426 const struct sockaddr_un* un = reinterpret_cast<const struct sockaddr_un*>(address);
427 const size_t pathLength = addressLength - pathOffset;
428 if (!pathLength) {
429 if (allowUnnamed) {
430 path = String();
431 return true;
432 }
433
434 SYSCALL_ERROR(InvalidArgument);
435 return false;
436 }
437
438 if (!un->sun_path[0]) {
439 static constexpr char digits[] = "0123456789abcdef";
440 char encoded[1 + 2 * sizeof(un->sun_path) + 1];
441 size_t encodedLength = 1;
442 encoded[0] = '\1';
443 for (size_t i = 1; i < pathLength; ++i) {
444 const uint8_t value = static_cast<uint8_t>(un->sun_path[i]);
445 encoded[encodedLength++] = digits[value >> 4];
446 encoded[encodedLength++] = digits[value & 0xf];
447 }
448 encoded[encodedLength] = 0;
449 path.assign(encoded, encodedLength);
450 return true;
451 }
452
453 char boundedPath[sizeof(un->sun_path) + 1];
454 ByteSet(boundedPath, 0, sizeof(boundedPath));
455 MemoryCopy(boundedPath, un->sun_path, pathLength);
456 normalisePath(path, boundedPath);
457 if (path.length() >= sizeof(un->sun_path)) {
458 SYSCALL_ERROR(NameTooLong);
459 return false;
460 }
461 return true;
462}
463
464static bool isAbstractUnixSocket(const String& address) {
465 return address.length() && address[0] == '\1';
466}
467
468static uint8_t decodeHexDigit(char value) {
469 return value >= 'a' ? static_cast<uint8_t>(value - 'a' + 10) : static_cast<uint8_t>(value - '0');
470}
471
472static void writeUnixSocketAddress(const String& value, struct sockaddr_storage* address,
473 socklen_t* addressLength) {
474 const size_t pathOffset = offsetof(struct sockaddr_un, sun_path);
475 const bool abstract = isAbstractUnixSocket(value);
476 const size_t nameLength = abstract ? (value.length() - 1) / 2 : value.length();
477 const size_t required = pathOffset + (abstract ? 1 + nameLength
478 : nameLength ? nameLength + 1
479 : 0);
480 const size_t capacity = addressLength ? *addressLength : 0;
481
482 if (address && capacity) {
483 ByteSet(address, 0, capacity < sizeof(sockaddr_un) ? capacity : sizeof(sockaddr_un));
484 auto* un = reinterpret_cast<struct sockaddr_un*>(address);
485 if (capacity >= sizeof(sa_family_t)) {
486 un->sun_family = AF_UNIX;
487 }
488 if (capacity > pathOffset) {
489 const size_t available = capacity - pathOffset;
490 if (abstract) {
491 const size_t amount = nameLength < available - 1 ? nameLength : available - 1;
492 for (size_t i = 0; i < amount; ++i) {
493 un->sun_path[i + 1] = static_cast<char>((decodeHexDigit(value[1 + 2 * i]) << 4) |
494 decodeHexDigit(value[2 + 2 * i]));
495 }
496 } else if (nameLength) {
497 const size_t amount = nameLength < available - 1 ? nameLength : available - 1;
498 MemoryCopy(un->sun_path, value.cstr(), amount);
499 }
500 }
501 }
502 if (addressLength) {
503 *addressLength = required;
504 }
505}
506
507static uint8_t lwipSocketOption(int option) {
508 switch (option) {
509 case SO_REUSEADDR:
510 return SOF_REUSEADDR;
511 case SO_KEEPALIVE:
512 return SOF_KEEPALIVE;
513 case SO_BROADCAST:
514 return SOF_BROADCAST;
515 default:
516 return 0;
517 }
518}
519
520static int lwipErrorNumber(err_t error) {
521 const int result = err_to_errno(error);
522 return result < 0 ? Error::IoError : result;
523}
524
525static err_t sockaddrToIpaddr(const struct sockaddr_storage* saddr, uint16_t& port,
526 ip_addr_t* result, bool isbind = true) {
527 ByteSet(result, 0, sizeof(*result));
528
529 if (saddr->ss_family == AF_INET) {
530 const struct sockaddr_in* sin = reinterpret_cast<const struct sockaddr_in*>(saddr);
531 result->u_addr.ip4.addr = sin->sin_addr.s_addr;
532 result->type = IPADDR_TYPE_V4;
533
534 if (!isbind) {
535 // do some extra sanity checks for client connections
536 if (!sin->sin_addr.s_addr) {
537 // rebind to 127.0.0.1 (localhost)
538 result->u_addr.ip4.addr = HOST_TO_BIG32(INADDR_LOOPBACK);
539 }
540 }
541
542 port = BIG_TO_HOST16(sin->sin_port);
543
544 return ERR_OK;
545 } else {
546 ERROR("sockaddrToIpaddr: only AF_INET is supported at the moment.");
547 }
548
549 return ERR_VAL;
550}
551
552int posix_socket(int domain, int type, int protocol) {
553 N_NOTICE("socket(" << domain << ", " << type << ", " << protocol << ")");
554
555 int socketType = 0;
556 int flags = 0;
557 if (!splitSocketType(type, socketType, flags)) {
558 return -1;
559 }
560 NetworkSyscalls* syscalls;
561 const auto network = posix_sandbox_network(*Processor::information().getCurrentThread());
562
563 if (domain == AF_UNIX) {
564 if (socketType != SOCK_STREAM && socketType != SOCK_DGRAM && socketType != SOCK_SEQPACKET) {
565 SYSCALL_ERROR(OperationNotSupported);
566 return -1;
567 }
568 syscalls = new UnixSocketSyscalls(domain, socketType, protocol);
569 } else if (network) {
570 syscalls = posix_network_socket(domain, socketType, protocol, network);
571 if (!syscalls) {
572 return -1;
573 }
574 } else if (domain == 16) {
575 syscalls = new MqueueNetlinkSocket(socketType, protocol);
576 } else {
578 syscalls = new LwipSocketSyscalls(domain, socketType, protocol);
579 }
580
581 if (!syscalls->create()) {
582 delete syscalls;
583 return -1;
584 }
585
588 setSocketDescriptorFlags(f, flags);
589 DescriptorLease installed;
590 const size_t fd = installDescriptor(f, installed);
591 syscalls->associate(f);
592
593 N_NOTICE(" -> " << Dec << fd << Hex);
594 return static_cast<int>(fd);
595}
596
597int posix_socketpair(int domain, int type, int protocol, int sv[2]) {
598 N_NOTICE("socketpair");
599
600 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(sv), sizeof(int) * 2,
601 PosixSubsystem::SafeWrite)) {
602 N_NOTICE("socketpair -> invalid address");
603 SYSCALL_ERROR(BadAddress);
604 return -1;
605 }
606
607 if (domain != AF_UNIX) {
608 N_NOTICE(" -> bad domain");
609 syscallError(EAFNOSUPPORT);
610 return -1;
611 }
612
613 int socketType = 0;
614 int flags = 0;
615 if (!splitSocketType(type, socketType, flags)) {
616 return -1;
617 }
618 if (socketType != SOCK_STREAM && socketType != SOCK_SEQPACKET) {
619 SYSCALL_ERROR(OperationNotSupported);
620 return -1;
621 }
622
623 UnixSocketSyscalls* syscallsA = new UnixSocketSyscalls(domain, socketType, protocol);
624 if (!syscallsA->create()) {
625 delete syscallsA;
626 N_NOTICE(" -> failed to create first socket");
627 return -1;
628 }
629
630 UnixSocketSyscalls* syscallsB = new UnixSocketSyscalls(domain, socketType, protocol);
631 if (!syscallsB->create()) {
632 delete syscallsA;
633 delete syscallsB;
634 N_NOTICE(" -> failed to create second socket");
635 return -1;
636 }
637
638 if (!syscallsA->pairWith(syscallsB)) {
639 delete syscallsA;
640 delete syscallsB;
641 N_NOTICE(" -> failed to pair");
642 return -1;
643 }
644
647
650
651 setSocketDescriptorFlags(fA, flags);
652 setSocketDescriptorFlags(fB, flags);
653
654 DescriptorLease installedA;
655 DescriptorLease installedB;
656 const size_t fdA = installDescriptor(fA, installedA);
657 const size_t fdB = installDescriptor(fB, installedB);
658
659 syscallsA->associate(fA);
660 syscallsB->associate(fB);
661
662 const int result[2] = {static_cast<int>(fdA), static_cast<int>(fdB)};
663 if (!PosixSubsystem::copyToUser(sv, result, sizeof(result))) {
664 removeDescriptor(result[0], installedA);
665 removeDescriptor(result[1], installedB);
666 SYSCALL_ERROR(BadAddress);
667 return -1;
668 }
669
670 N_NOTICE(" -> " << result[0] << ", " << result[1]);
671 return 0;
672}
673
674int posix_connect(int sock, const struct sockaddr_storage* address, socklen_t addrlen) {
675 N_NOTICE("connect");
676
677 struct sockaddr_storage snapshot = {};
678 if (!copySocketAddress(address, addrlen, snapshot)) {
679 return -1;
680 }
681
682 N_NOTICE("connect(" << sock << ", " << reinterpret_cast<uintptr_t>(address) << ", " << addrlen
683 << ")");
684
686 acquireDescriptor(sock, f);
687 if (!isSaneSocket(f)) {
688 return -1;
689 }
690
691 if (snapshot.ss_family != f->networkImpl->getDomain()) {
692 syscallError(EAFNOSUPPORT);
693 N_NOTICE(" -> incorrect address family passed to connect()");
694 return -1;
695 }
696
697 Thread* thread = beginInterruptibleSocketCall();
698 const int result = f->networkImpl->connect(&snapshot, addrlen);
699 return finishInterruptibleSocketCall(thread, static_cast<ssize_t>(result)) ? result : -1;
700}
701
702ssize_t posix_send(int sock, const void* buff, size_t bufflen, int flags) {
703 N_NOTICE("send");
704
705 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(buff), bufflen,
706 PosixSubsystem::SafeRead)) {
707 N_NOTICE("send -> invalid address");
708 SYSCALL_ERROR(BadAddress);
709 return -1;
710 }
711
712 N_NOTICE("send(" << sock << ", " << buff << ", " << bufflen << ", " << flags << ")");
713
715 acquireDescriptor(sock, f);
716 return posix_send_descriptor(f, buff, bufflen, flags);
717}
718
719ssize_t posix_send_descriptor(const DescriptorLease& f, const void* buff, size_t bufflen, int flags,
720 bool kernelBuffer) {
721 if (!isSaneSocket(f)) {
722 return -1;
723 }
724
725 struct iovec vector = {const_cast<void*>(buff), bufflen};
726 struct msghdr message = {};
727 message.msg_iov = &vector;
728 message.msg_iovlen = 1;
729 message.msg_flags = flags;
730 return posix_sendmsg_descriptor(f, &message, SharedPointer<SocketRights>(), kernelBuffer);
731}
732
733ssize_t posix_sendmsg_descriptor(const DescriptorLease& f, const struct msghdr* message,
734 const SharedPointer<SocketRights>& rights, bool kernelBuffer) {
735 if (!isSaneSocket(f) ||
736 !validateSocketMessageFlags(message->msg_flags, true, f->networkImpl->getDomain(),
737 f->networkImpl->getType())) {
738 return -1;
739 }
740
741 SocketPayload payload;
742 if (!payload.prepare(*message, f->networkImpl->getType(), f->networkImpl->getDomain(), true,
743 kernelBuffer)) {
744 return -1;
745 }
746 struct msghdr snapshot = *message;
747 payload.attach(snapshot);
748 Thread* thread = beginInterruptibleSocketCall();
749 const ssize_t result = f->networkImpl->sendto_msg(&snapshot, rights);
750 return finishInterruptibleSocketCall(thread, result) ? result : -1;
751}
752
753ssize_t posix_sendto(int sock, const void* buff, size_t bufflen, int flags,
754 struct sockaddr_storage* address, socklen_t addrlen) {
755 N_NOTICE("sendto");
756
757 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(buff), bufflen,
758 PosixSubsystem::SafeRead)) {
759 N_NOTICE("sendto -> invalid address for transmission buffer");
760 SYSCALL_ERROR(BadAddress);
761 return -1;
762 }
763 struct sockaddr_storage destination = {};
764 const struct sockaddr_storage* destinationAddress = nullptr;
765 if (address) {
766 if (addrlen < sizeof(sa_family_t) || addrlen > sizeof(destination)) {
767 N_NOTICE("sendto -> invalid destination address length");
768 SYSCALL_ERROR(InvalidArgument);
769 return -1;
770 }
771 if (!PosixSubsystem::copyFromUser(&destination, address, addrlen)) {
772 N_NOTICE("sendto -> invalid destination address");
773 SYSCALL_ERROR(BadAddress);
774 return -1;
775 }
776 destinationAddress = &destination;
777 }
778
779 N_NOTICE("sendto(" << sock << ", " << buff << ", " << bufflen << ", " << flags << ", " << address
780 << ", " << addrlen << ")");
781
783 acquireDescriptor(sock, f);
784 if (!isSaneSocket(f)) {
785 return -1;
786 }
787
788 struct iovec vector = {const_cast<void*>(buff), bufflen};
789 struct msghdr message = {};
790 message.msg_name = const_cast<struct sockaddr_storage*>(destinationAddress);
791 message.msg_namelen = addrlen;
792 message.msg_iov = &vector;
793 message.msg_iovlen = 1;
794 message.msg_flags = flags;
795 return posix_sendmsg_descriptor(f, &message);
796}
797
798ssize_t posix_recv(int sock, void* buff, size_t bufflen, int flags) {
799 N_NOTICE("recv");
800
801 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(buff), bufflen,
802 PosixSubsystem::SafeWrite)) {
803 N_NOTICE("recv -> invalid address");
804 SYSCALL_ERROR(BadAddress);
805 return -1;
806 }
807
808 N_NOTICE("recv(" << sock << ", " << buff << ", " << bufflen << ", " << flags << ")");
809
811 acquireDescriptor(sock, f);
812 ssize_t n = posix_recv_descriptor(f, buff, bufflen, flags);
813
814 N_NOTICE(" -> " << n);
815 return n;
816}
817
818ssize_t posix_recv_descriptor(const DescriptorLease& f, void* buff, size_t bufflen, int flags) {
819 if (!isSaneSocket(f)) {
820 return -1;
821 }
822 if (!validateSocketMessageFlags(flags, false, f->networkImpl->getDomain())) {
823 return -1;
824 }
825
826 struct iovec vector = {buff, bufflen};
827 struct msghdr message = {};
828 message.msg_iov = &vector;
829 message.msg_iovlen = 1;
830 message.msg_flags = flags;
831 return posix_recvmsg_descriptor(f, &message);
832}
833
834ssize_t posix_recvmsg_descriptor(const DescriptorLease& f, struct msghdr* message,
836 if (!isSaneSocket(f)) {
837 return -1;
838 }
839
840 const int pendingError = f->networkImpl->takeReceiveError();
841 if (pendingError) {
842 syscallError(pendingError);
843 return -1;
844 }
845 SocketPayload payload;
846 if (!payload.prepare(*message, f->networkImpl->getType(), f->networkImpl->getDomain(), false)) {
847 return -1;
848 }
849 struct msghdr snapshot = *message;
850 payload.attach(snapshot);
851 Thread* thread = beginInterruptibleSocketCall();
852 const ssize_t result = f->networkImpl->recvfrom_msg(&snapshot, rights);
853 if (!finishInterruptibleSocketCall(thread, result) ||
854 (result >= 0 && !payload.copyReceived(*message, static_cast<size_t>(result)))) {
855 if (rights) {
856 rights->reset();
857 }
858 return -1;
859 }
860 message->msg_namelen = snapshot.msg_namelen;
861 message->msg_controllen = snapshot.msg_controllen;
862 message->msg_flags = snapshot.msg_flags;
863 return result;
864}
865
866ssize_t posix_recvfrom(int sock, void* buff, size_t bufflen, int flags,
867 struct sockaddr_storage* address, socklen_t* addrlen) {
868 N_NOTICE("recvfrom");
869
871 acquireDescriptor(sock, f);
872 if (!isSaneSocket(f) || !validateSocketMessageFlags(flags, false, f->networkImpl->getDomain())) {
873 return -1;
874 }
875
876 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(buff), bufflen,
877 PosixSubsystem::SafeWrite)) {
878 N_NOTICE("recvfrom -> invalid receive buffer");
879 SYSCALL_ERROR(BadAddress);
880 return -1;
881 }
882
883 struct sockaddr_storage source = {};
884 struct sockaddr_storage* sourceAddress = nullptr;
885 socklen_t sourceCapacity = 0;
886 if (address) {
887 if (!PosixSubsystem::copyFromUser(&sourceCapacity, addrlen, sizeof(sourceCapacity))) {
888 N_NOTICE("recvfrom -> invalid source address length");
889 SYSCALL_ERROR(BadAddress);
890 return -1;
891 }
892 if (sourceCapacity > static_cast<socklen_t>(INT_MAX)) {
893 SYSCALL_ERROR(InvalidArgument);
894 return -1;
895 }
896
897 const size_t checkedCapacity =
898 sourceCapacity < sizeof(source) ? sourceCapacity : sizeof(source);
899 if (checkedCapacity &&
900 !PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(address), checkedCapacity,
901 PosixSubsystem::SafeWrite)) {
902 N_NOTICE("recvfrom -> invalid source address buffer");
903 SYSCALL_ERROR(BadAddress);
904 return -1;
905 }
906 sourceAddress = &source;
907 }
908
909 N_NOTICE("recvfrom(" << sock << ", " << buff << ", " << bufflen << ", " << flags << ", "
910 << address << ", " << addrlen);
911
912 struct iovec vector = {buff, bufflen};
913 struct msghdr message = {};
914 message.msg_iov = &vector;
915 message.msg_iovlen = 1;
916 message.msg_flags = flags;
917 message.msg_name = sourceAddress;
918 message.msg_namelen = sourceAddress ? sizeof(source) : 0;
919 ssize_t n = posix_recvmsg_descriptor(f, &message);
920 const socklen_t sourceLength = message.msg_namelen;
921
922 if (n >= 0 && sourceAddress) {
923 size_t copyLength = sourceLength;
924 if (copyLength > sourceCapacity) {
925 copyLength = sourceCapacity;
926 }
927 if (copyLength > sizeof(source)) {
928 copyLength = sizeof(source);
929 }
930 if ((copyLength && !PosixSubsystem::copyToUser(address, &source, copyLength)) ||
931 !PosixSubsystem::copyToUser(addrlen, &sourceLength, sizeof(sourceLength))) {
932 SYSCALL_ERROR(BadAddress);
933 return -1;
934 }
935 }
936
937 N_NOTICE(" -> " << n);
938 return n;
939}
940
941int posix_bind(int sock, const struct sockaddr_storage* address, socklen_t addrlen) {
942 N_NOTICE("bind");
943
944 struct sockaddr_storage snapshot = {};
945 if (!copySocketAddress(address, addrlen, snapshot)) {
946 return -1;
947 }
948
949 N_NOTICE("bind(" << sock << ", " << address << ", " << addrlen << ")");
950
952 acquireDescriptor(sock, f);
953 if (!isSaneSocket(f)) {
954 return -1;
955 }
956
957 if (f->networkImpl->getDomain() != snapshot.ss_family) {
958 syscallError(EAFNOSUPPORT);
959 return -1;
960 }
961
962 return f->networkImpl->bind(&snapshot, addrlen);
963}
964
965int posix_listen(int sock, int backlog) {
966 N_NOTICE("listen(" << sock << ", " << backlog << ")");
967
969 acquireDescriptor(sock, f);
970 if (!isSaneSocket(f)) {
971 return -1;
972 }
973
974 if (f->networkImpl->getType() != SOCK_STREAM &&
975 !(f->networkImpl->getDomain() == AF_UNIX && f->networkImpl->getType() == SOCK_SEQPACKET)) {
976 SYSCALL_ERROR(InvalidArgument);
977 return -1;
978 }
979
980 return f->networkImpl->listen(backlog);
981}
982
983int posix_accept(int sock, struct sockaddr_storage* address, socklen_t* addrlen) {
984 return posix_accept4(sock, address, addrlen, 0);
985}
986
987int posix_accept4(int sock, struct sockaddr_storage* address, socklen_t* addrlen, int flags) {
988 N_NOTICE("accept4");
989
990 if (flags & ~socketCreationFlags) {
991 SYSCALL_ERROR(InvalidArgument);
992 return -1;
993 }
994
995 struct sockaddr_storage acceptedAddress;
996 ByteSet(&acceptedAddress, 0, sizeof(acceptedAddress));
997 socklen_t acceptedLength = sizeof(acceptedAddress);
998 socklen_t addressCapacity = 0;
999 const bool returnAddress = address != nullptr;
1000 if (returnAddress) {
1001 if (!PosixSubsystem::copyFromUser(&addressCapacity, addrlen, sizeof(addressCapacity)) ||
1002 !PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(addrlen), sizeof(socklen_t),
1003 PosixSubsystem::SafeWrite)) {
1004 N_NOTICE("accept4 -> invalid address length");
1005 SYSCALL_ERROR(BadAddress);
1006 return -1;
1007 }
1008 if (addressCapacity > static_cast<socklen_t>(INT_MAX)) {
1009 SYSCALL_ERROR(InvalidArgument);
1010 return -1;
1011 }
1012
1013 const size_t writableLength = addressCapacity < sizeof(acceptedAddress)
1014 ? static_cast<size_t>(addressCapacity)
1015 : sizeof(acceptedAddress);
1016 if (writableLength &&
1017 !PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(address), writableLength,
1018 PosixSubsystem::SafeWrite)) {
1019 N_NOTICE("accept4 -> invalid address");
1020 SYSCALL_ERROR(BadAddress);
1021 return -1;
1022 }
1023 }
1024
1025 N_NOTICE("accept4(" << sock << ", " << address << ", " << addrlen << ", " << flags << ")");
1026
1028 acquireDescriptor(sock, f);
1029 if (!isSaneSocket(f)) {
1030 return -1;
1031 }
1032
1033 if (f->networkImpl->getType() != SOCK_STREAM &&
1034 !(f->networkImpl->getDomain() == AF_UNIX && f->networkImpl->getType() == SOCK_SEQPACKET)) {
1035 SYSCALL_ERROR(OperationNotSupported);
1036 return -1;
1037 }
1038
1039 Thread* thread = beginInterruptibleSocketCall();
1040 DescriptorLease accepted;
1041 int r = f->networkImpl->accept(&acceptedAddress, &acceptedLength, flags, &accepted);
1042 if (!finishInterruptibleSocketCall(thread, static_cast<ssize_t>(r))) {
1043 return -1;
1044 }
1045 if (r >= 0 && returnAddress) {
1046 const size_t copyLength = addressCapacity < acceptedLength
1047 ? static_cast<size_t>(addressCapacity)
1048 : static_cast<size_t>(acceptedLength);
1049 if (acceptedLength > sizeof(acceptedAddress) ||
1050 !PosixSubsystem::copyToUser(address, &acceptedAddress, copyLength) ||
1051 !PosixSubsystem::copyToUser(addrlen, &acceptedLength, sizeof(acceptedLength))) {
1052 removeDescriptor(r, accepted);
1053 SYSCALL_ERROR(BadAddress);
1054 return -1;
1055 }
1056 }
1057 N_NOTICE(" -> " << Dec << r);
1058 return r;
1059}
1060
1061int posix_shutdown(int socket, int how) {
1062 N_NOTICE("shutdown(" << socket << ", " << how << ")");
1063
1065 acquireDescriptor(socket, f);
1066 if (!isSaneSocket(f)) {
1067 return -1;
1068 }
1069
1070 return f->networkImpl->shutdown(how);
1071}
1072
1073namespace {
1074int socketName(int socket, struct sockaddr_storage* address, socklen_t* addressLength, bool peer) {
1075 socklen_t capacity = 0;
1076 if (!PosixSubsystem::copyFromUser(&capacity, addressLength, sizeof(capacity))) {
1077 SYSCALL_ERROR(BadAddress);
1078 return -1;
1079 }
1080 if (capacity > INT_MAX) {
1081 SYSCALL_ERROR(InvalidArgument);
1082 return -1;
1083 }
1084
1086 acquireDescriptor(socket, f);
1087 if (!isSaneSocket(f)) {
1088 return -1;
1089 }
1090
1091 struct sockaddr_storage result = {};
1092 socklen_t length = sizeof(result);
1093 const int status = peer ? f->networkImpl->getpeername(&result, &length)
1094 : f->networkImpl->getsockname(&result, &length);
1095 if (status < 0) {
1096 return -1;
1097 }
1098 if (length > sizeof(result)) {
1099 SYSCALL_ERROR(IoError);
1100 return -1;
1101 }
1102
1103 const size_t copied = capacity < length ? capacity : length;
1104 if (!PosixSubsystem::copyToUser(address, &result, copied) ||
1105 !PosixSubsystem::copyToUser(addressLength, &length, sizeof(length))) {
1106 SYSCALL_ERROR(BadAddress);
1107 return -1;
1108 }
1109 return 0;
1110}
1111} // namespace
1112
1113int posix_getpeername(int socket, struct sockaddr_storage* address, socklen_t* address_len) {
1114 return socketName(socket, address, address_len, true);
1115}
1116
1117int posix_getsockname(int socket, struct sockaddr_storage* address, socklen_t* address_len) {
1118 return socketName(socket, address, address_len, false);
1119}
1120
1121int posix_setsockopt(int sock, int level, int optname, const void* optvalue, socklen_t optlen) {
1122 if (optlen < sizeof(int)) {
1123 SYSCALL_ERROR(InvalidArgument);
1124 return -1;
1125 }
1126 int value = 0;
1127 if (!PosixSubsystem::copyFromUser(&value, optvalue, sizeof(value))) {
1128 SYSCALL_ERROR(BadAddress);
1129 return -1;
1130 }
1132 acquireDescriptor(sock, f);
1133 if (!isSaneSocket(f)) {
1134 return -1;
1135 }
1136 return f->networkImpl->setsockopt(level, optname, &value, sizeof(value));
1137}
1138
1139int posix_getsockopt(int sock, int level, int optname, void* optvalue, socklen_t* optlen) {
1140 socklen_t capacity = 0;
1141 if (!PosixSubsystem::copyFromUser(&capacity, optlen, sizeof(capacity))) {
1142 SYSCALL_ERROR(BadAddress);
1143 return -1;
1144 }
1145 if (capacity > INT_MAX) {
1146 SYSCALL_ERROR(InvalidArgument);
1147 return -1;
1148 }
1149 union {
1150 int scalar;
1151 struct ucred credentials;
1152 } value = {};
1153 socklen_t length = sizeof(value);
1155 acquireDescriptor(sock, f);
1156 if (!isSaneSocket(f)) {
1157 return -1;
1158 }
1159 const int pendingError =
1160 level == SOL_SOCKET && optname == SO_ERROR ? f->networkImpl->takeReceiveError() : 0;
1161 if (pendingError) {
1162 value.scalar = pendingError;
1163 length = sizeof(value.scalar);
1164 } else if (f->networkImpl->getsockopt(level, optname, &value, &length) < 0) {
1165 return -1;
1166 }
1167 if (length > sizeof(value)) {
1168 SYSCALL_ERROR(IoError);
1169 return -1;
1170 }
1171 const socklen_t copied = capacity < length ? capacity : length;
1172 if (!PosixSubsystem::copyToUser(optvalue, &value, copied) ||
1173 !PosixSubsystem::copyToUser(optlen, &copied, sizeof(copied))) {
1174 SYSCALL_ERROR(BadAddress);
1175 return -1;
1176 }
1177 return 0;
1178}
1179
1180ssize_t posix_sendmsg(int sockfd, const struct msghdr* msg, int flags) {
1181 N_NOTICE("sendmsg(" << sockfd << ", " << msg << ", " << flags << ")");
1182
1183 struct msghdr message = {};
1184 if (!PosixSubsystem::copyFromUser(&message, msg, sizeof(message))) {
1185 SYSCALL_ERROR(BadAddress);
1186 return -1;
1187 }
1188
1190 if (!parseSocketRights(message, rights)) {
1191 return -1;
1192 }
1193
1194 constexpr size_t MaximumIoVectors = 1024;
1195 const size_t vectorCount = static_cast<size_t>(message.msg_iovlen);
1196 if (vectorCount > MaximumIoVectors) {
1197 SYSCALL_ERROR(InvalidArgument);
1198 return -1;
1199 }
1200
1201 UniqueArray<struct iovec> vectorOwner;
1202 if (vectorCount) {
1203 vectorOwner = UniqueArray<struct iovec>::allocate(vectorCount);
1204 }
1205 struct iovec* vectors = vectorOwner.get();
1206 if (vectorCount &&
1207 !PosixSubsystem::copyFromUser(vectors, message.msg_iov, vectorCount, sizeof(*vectors))) {
1208 SYSCALL_ERROR(BadAddress);
1209 return -1;
1210 }
1211 constexpr size_t MaximumIoBytes = static_cast<size_t>(SSIZE_MAX);
1212 size_t totalLength = 0;
1213 for (size_t i = 0; i < vectorCount; ++i) {
1214 if (vectors[i].iov_len > MaximumIoBytes - totalLength) {
1215 SYSCALL_ERROR(InvalidArgument);
1216 return -1;
1217 }
1218 if (!PosixSubsystem::checkUserBuffer(reinterpret_cast<uintptr_t>(vectors[i].iov_base),
1219 vectors[i].iov_len, 1, PosixSubsystem::SafeRead)) {
1220 SYSCALL_ERROR(BadAddress);
1221 return -1;
1222 }
1223 totalLength += vectors[i].iov_len;
1224 }
1225
1226 struct sockaddr_storage address = {};
1227 if (message.msg_name) {
1228 if (message.msg_namelen > sizeof(address)) {
1229 SYSCALL_ERROR(InvalidArgument);
1230 return -1;
1231 }
1232 if (!PosixSubsystem::copyFromUser(&address, message.msg_name, message.msg_namelen)) {
1233 SYSCALL_ERROR(BadAddress);
1234 return -1;
1235 }
1236 message.msg_name = &address;
1237 }
1238 message.msg_iov = vectors;
1239 message.msg_control = nullptr;
1240 message.msg_controllen = 0;
1241 message.msg_flags = flags;
1242
1244 acquireDescriptor(sockfd, f);
1245 if (!isSaneSocket(f)) {
1246 return -1;
1247 }
1248 if (rights &&
1249 (f->networkImpl->getDomain() != AF_UNIX ||
1250 (f->networkImpl->getType() != SOCK_DGRAM && f->networkImpl->getType() != SOCK_STREAM &&
1251 f->networkImpl->getType() != SOCK_SEQPACKET))) {
1252 SYSCALL_ERROR(OperationNotSupported);
1253 return -1;
1254 }
1255 if (rights && f->networkImpl->getType() == SOCK_STREAM && !totalLength) {
1256 SYSCALL_ERROR(InvalidArgument);
1257 return -1;
1258 }
1259
1260 const ssize_t n = posix_sendmsg_descriptor(f, &message, rights);
1261 N_NOTICE(" -> " << n);
1262 return n;
1263}
1264
1265ssize_t posix_recvmsg(int sockfd, struct msghdr* msg, int flags) {
1266 N_NOTICE("recvmsg(" << sockfd << ", " << msg << ", " << flags << ")");
1267
1269 acquireDescriptor(sockfd, f);
1270 if (!isSaneSocket(f) || !validateSocketMessageFlags(flags, false, f->networkImpl->getDomain())) {
1271 return -1;
1272 }
1273
1274 return posix_recvmsg_user_descriptor(f, msg, flags);
1275}
1276
1277ssize_t posix_recvmsg_user_descriptor(const DescriptorLease& f, struct msghdr* msg, int flags,
1278 unsigned int* receivedLength) {
1279 if (!isSaneSocket(f) || !validateSocketMessageFlags(flags, false, f->networkImpl->getDomain()))
1280 return -1;
1281 struct msghdr message = {};
1282 if (!PosixSubsystem::copyFromUser(&message, msg, sizeof(message)) ||
1283 !PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(msg), sizeof(message),
1284 PosixSubsystem::SafeWrite)) {
1285 SYSCALL_ERROR(BadAddress);
1286 return -1;
1287 }
1288 if (receivedLength &&
1289 !PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(receivedLength),
1290 sizeof(*receivedLength), PosixSubsystem::SafeWrite)) {
1291 SYSCALL_ERROR(BadAddress);
1292 return -1;
1293 }
1294 const struct msghdr originalMessage = message;
1295
1296 constexpr size_t MaximumIoVectors = 1024;
1297 const size_t vectorCount = static_cast<size_t>(message.msg_iovlen);
1298 if (vectorCount > MaximumIoVectors) {
1299 SYSCALL_ERROR(InvalidArgument);
1300 return -1;
1301 }
1302
1303 UniqueArray<struct iovec> vectorOwner;
1304 if (vectorCount) {
1305 vectorOwner = UniqueArray<struct iovec>::allocate(vectorCount);
1306 }
1307 struct iovec* vectors = vectorOwner.get();
1308 if (vectorCount &&
1309 !PosixSubsystem::copyFromUser(vectors, message.msg_iov, vectorCount, sizeof(*vectors))) {
1310 SYSCALL_ERROR(BadAddress);
1311 return -1;
1312 }
1313 constexpr size_t MaximumIoBytes = static_cast<size_t>(SSIZE_MAX);
1314 size_t totalLength = 0;
1315 for (size_t i = 0; i < vectorCount; ++i) {
1316 if (vectors[i].iov_len > MaximumIoBytes - totalLength) {
1317 SYSCALL_ERROR(InvalidArgument);
1318 return -1;
1319 }
1320 if (!PosixSubsystem::checkUserBuffer(reinterpret_cast<uintptr_t>(vectors[i].iov_base),
1321 vectors[i].iov_len, 1, PosixSubsystem::SafeWrite)) {
1322 SYSCALL_ERROR(BadAddress);
1323 return -1;
1324 }
1325 totalLength += vectors[i].iov_len;
1326 }
1327
1328 void* userName = message.msg_name;
1329 const size_t userNameCapacity = message.msg_namelen;
1330 void* userControl = message.msg_control;
1331 const size_t userControlCapacity = static_cast<size_t>(message.msg_controllen);
1332 struct sockaddr_storage address = {};
1333 if (userName) {
1334 const size_t checkedCapacity =
1335 userNameCapacity < sizeof(address) ? userNameCapacity : sizeof(address);
1336 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(userName), checkedCapacity,
1337 PosixSubsystem::SafeWrite)) {
1338 SYSCALL_ERROR(BadAddress);
1339 return -1;
1340 }
1341 message.msg_name = &address;
1342 message.msg_namelen = checkedCapacity;
1343 }
1344 if (userControlCapacity) {
1345 if (!userControl) {
1346 SYSCALL_ERROR(BadAddress);
1347 return -1;
1348 }
1349 const size_t checkedCapacity =
1350 userControlCapacity < MaximumControlBytes ? userControlCapacity : MaximumControlBytes;
1351 if (!PosixSubsystem::checkAddress(reinterpret_cast<uintptr_t>(userControl), checkedCapacity,
1352 PosixSubsystem::SafeWrite)) {
1353 SYSCALL_ERROR(BadAddress);
1354 return -1;
1355 }
1356 }
1357 message.msg_iov = vectors;
1358 message.msg_control = nullptr;
1359 message.msg_controllen = 0;
1360 message.msg_flags = flags;
1361
1363 const ssize_t n = posix_recvmsg_descriptor(f, &message, &rights);
1364
1365 if (n >= 0) {
1366 struct msghdr result = originalMessage;
1367 const size_t rightsCount = rights ? rights->count() : 0;
1368 size_t disclosedCount = 0;
1369 if (rightsCount && userControl && userControlCapacity >= CMSG_LEN(sizeof(int))) {
1370 disclosedCount = (userControlCapacity - CMSG_LEN(0)) / sizeof(int);
1371 if (disclosedCount > rightsCount) {
1372 disclosedCount = rightsCount;
1373 }
1374 }
1375
1376 UniqueArray<int> installedFds;
1377 UniqueArray<DescriptorLease> installedLeases;
1378 if (disclosedCount) {
1379 installedFds = UniqueArray<int>::allocate(disclosedCount);
1380 installedLeases = UniqueArray<DescriptorLease>::allocate(disclosedCount);
1381 }
1382
1383 PosixSubsystem* subsystem = getSubsystem();
1384 size_t publishedCount = 0;
1385 auto rollback = [&]() {
1386 for (size_t i = 0; i < publishedCount; ++i) {
1387 subsystem->closeFileDescriptor(static_cast<size_t>(installedFds.get()[i]),
1388 installedLeases.get()[i]);
1389 installedLeases.get()[i].reset();
1390 }
1391 };
1392
1393 for (; publishedCount < disclosedCount; ++publishedCount) {
1394 FileDescriptor* received = new FileDescriptor(*rights->descriptor(publishedCount));
1395 if ((rights->descriptor(publishedCount)->networkImpl && !received->networkPublished()) ||
1396 (rights->descriptor(publishedCount)->getEventFdImpl() && !received->eventFdPublished()) ||
1397 (rights->descriptor(publishedCount)->getTimerFdImpl() && !received->timerFdPublished()) ||
1398 (rights->descriptor(publishedCount)->getSignalFdImpl() &&
1399 !received->signalFdPublished())) {
1400 delete received;
1401 rollback();
1402 SYSCALL_ERROR(BadFileDescriptor);
1403 return -1;
1404 }
1405 int descriptorFlags = 0;
1406#ifdef MSG_CMSG_CLOEXEC
1407 descriptorFlags = (flags & MSG_CMSG_CLOEXEC) ? FD_CLOEXEC : 0;
1408#endif
1409 received->setFlags(descriptorFlags);
1410 const size_t fd =
1411 subsystem->installFileDescriptor(received, installedLeases.get()[publishedCount]);
1412 installedFds.get()[publishedCount] = static_cast<int>(fd);
1413 }
1414
1415 size_t controlBytes = 0;
1416 UniqueArray<uint8_t> control;
1417 if (disclosedCount) {
1418 const size_t fullSpace = CMSG_SPACE(disclosedCount * sizeof(int));
1419 controlBytes = userControlCapacity < fullSpace ? userControlCapacity : fullSpace;
1420 control = UniqueArray<uint8_t>::allocate(controlBytes);
1421 ByteSet(control.get(), 0, controlBytes);
1422
1423 struct cmsghdr header = {};
1424 header.cmsg_len = CMSG_LEN(disclosedCount * sizeof(int));
1425 header.cmsg_level = SOL_SOCKET;
1426 header.cmsg_type = SCM_RIGHTS;
1427 MemoryCopy(control.get(), &header, sizeof(header));
1428 MemoryCopy(control.get() + CMSG_LEN(0), installedFds.get(), disclosedCount * sizeof(int));
1429 }
1430
1431 if (userName) {
1432 size_t nameBytes = message.msg_namelen;
1433 if (nameBytes > userNameCapacity) {
1434 nameBytes = userNameCapacity;
1435 }
1436 if (nameBytes > sizeof(address)) {
1437 nameBytes = sizeof(address);
1438 }
1439 if (nameBytes && !PosixSubsystem::copyToUser(userName, &address, nameBytes)) {
1440 rollback();
1441 SYSCALL_ERROR(BadAddress);
1442 return -1;
1443 }
1444 }
1445
1446 if (controlBytes && !PosixSubsystem::copyToUser(userControl, control.get(), controlBytes)) {
1447 rollback();
1448 SYSCALL_ERROR(BadAddress);
1449 return -1;
1450 }
1451
1452 result.msg_namelen = message.msg_namelen;
1453 result.msg_controllen = controlBytes;
1454 result.msg_flags = message.msg_flags;
1455#ifdef MSG_CTRUNC
1456 if (disclosedCount < rightsCount) {
1457 result.msg_flags |= MSG_CTRUNC;
1458 }
1459#endif
1460 const unsigned int length = static_cast<unsigned int>(n);
1461 if ((receivedLength && !PosixSubsystem::copyToUser(receivedLength, &length, sizeof(length))) ||
1462 !PosixSubsystem::copyToUser(msg, &result, sizeof(result))) {
1463 rollback();
1464 SYSCALL_ERROR(BadAddress);
1465 return -1;
1466 }
1467 }
1468 N_NOTICE(" -> " << n);
1469 return n;
1470}
1471
1472NetworkSyscalls::NetworkSyscalls(int domain, int type, int protocol)
1473 : m_Domain(domain),
1474 m_Type(type),
1475 m_Protocol(protocol),
1476 m_NetworkNamespace(Processor::information().getCurrentThread()
1477 ? posix_sandbox_network(*Processor::information().getCurrentThread())
1479 m_Blocking(true),
1480 m_ReadinessNotifications(),
1481 m_LifecycleLock(),
1482 m_DescriptorOwners(0),
1483 m_DescriptorAdmissionOpen(true),
1484 m_LastDescriptorClosed(false),
1485 m_DescriptorLifetime() {}
1486
1487int NetworkSyscalls::takeReceiveError() {
1488 int error;
1489 {
1490 ConstexprLockGuard<Mutex, THREADS> guard(m_ReceiveErrorLock);
1491 error = m_ReceiveError;
1492 m_ReceiveError = 0;
1493 }
1494 if (error)
1495 notifyReadiness(ReadyError);
1496 return error;
1497}
1498
1499void NetworkSyscalls::deferReceiveError(int error) {
1500 {
1501 ConstexprLockGuard<Mutex, THREADS> guard(m_ReceiveErrorLock);
1502 if (!m_ReceiveError && error)
1503 ++m_ReceiveErrorGeneration;
1504 m_ReceiveError = error;
1505 }
1506 notifyReadiness(ReadyError);
1507}
1508
1509ReadyMask NetworkSyscalls::pendingReceiveReadiness() const {
1510 ConstexprLockGuard<Mutex, THREADS> guard(m_ReceiveErrorLock);
1511 return m_ReceiveError ? ReadyError : ReadyNone;
1512}
1513
1514ReadinessGenerations NetworkSyscalls::withReceiveErrorGeneration(
1515 ReadinessGenerations generations) const {
1516 ConstexprLockGuard<Mutex, THREADS> guard(m_ReceiveErrorLock);
1517 // The independent error source must survive drain/refill callback reorder.
1518 generations.error += m_ReceiveErrorGeneration;
1519 return generations;
1520}
1521
1523 return withReceiveErrorGeneration(ReadinessGenerations());
1524}
1525
1526NetworkSyscalls::~NetworkSyscalls() {
1528}
1529
1530bool NetworkSyscalls::create() {
1531 return true;
1532}
1533
1534ssize_t NetworkSyscalls::sendto(const void* buffer, size_t bufferlen, int flags,
1535 const struct sockaddr_storage* address, socklen_t addrlen) {
1536 struct iovec iov;
1537 iov.iov_base = const_cast<void*>(buffer);
1538 iov.iov_len = bufferlen;
1539
1540 struct msghdr msg;
1541 msg.msg_name = const_cast<struct sockaddr_storage*>(address);
1542 msg.msg_namelen = addrlen;
1543 msg.msg_iov = &iov;
1544 msg.msg_iovlen = 1;
1545 msg.msg_control = nullptr;
1546 msg.msg_controllen = 0;
1547 msg.msg_flags = flags;
1548
1550 return sendto_msg(&msg, rights);
1551}
1552
1553ssize_t NetworkSyscalls::recvfrom(void* buffer, size_t bufferlen, int flags,
1554 struct sockaddr_storage* address, socklen_t* addrlen) {
1555 struct iovec iov;
1556 iov.iov_base = buffer;
1557 iov.iov_len = bufferlen;
1558
1559 struct msghdr msg;
1560 msg.msg_name = address;
1561 msg.msg_namelen = addrlen ? *addrlen : 0;
1562 msg.msg_iov = &iov;
1563 msg.msg_iovlen = 1;
1564 msg.msg_control = nullptr;
1565 msg.msg_controllen = 0;
1566 msg.msg_flags = flags;
1567
1568 ssize_t result = recvfrom_msg(&msg, nullptr);
1569 if (result >= 0) {
1570 // Copy result address length if needed.
1571 if (addrlen) {
1572 *addrlen = msg.msg_namelen;
1573 }
1574 }
1575
1576 return result;
1577}
1578
1579int NetworkSyscalls::shutdown(int how) {
1580 return 0;
1581}
1582
1583bool NetworkSyscalls::canPoll() const {
1584 return false;
1585}
1586
1587bool NetworkSyscalls::poll(bool& read, bool& write, bool& error, Semaphore* waiter) {
1588 read = false;
1589 write = false;
1590 error = false;
1591 return false;
1592}
1593
1594void NetworkSyscalls::unPoll(Semaphore* waiter) {}
1595
1596ReadyMask NetworkSyscalls::queryReady(bool reading, bool writing) {
1597 (void)reading;
1598 (void)writing;
1599 return ReadyInvalid | pendingReceiveReadiness();
1600}
1601
1603 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1604 if (!m_DescriptorAdmissionOpen) {
1605 return false;
1606 }
1607
1608 ++m_DescriptorOwners;
1609 return true;
1610}
1611
1612void NetworkSyscalls::removeDescriptorOwner() {
1613 bool closeEndpoint = false;
1614 {
1615 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1616 assert(m_DescriptorOwners);
1617 --m_DescriptorOwners;
1618 if (!m_DescriptorOwners) {
1619 m_DescriptorAdmissionOpen = false;
1620 // AF_UNIX blocking operations use generation-held stream storage, so
1621 // retiring the table-visible endpoint wakes them safely. lwIP calls may
1622 // still be using the netconn through a syscall lease and retire when
1623 // that final object reference drains instead.
1624 closeEndpoint = usesLocalEndpointLifetime() || m_Domain == 16;
1625 }
1626 }
1627
1628 if (closeEndpoint) {
1630 }
1631}
1632
1634 if (!usesLocalEndpointLifetime() || !lifetime) {
1635 return;
1636 }
1637
1638 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1639 if (!m_DescriptorLifetime) {
1640 m_DescriptorLifetime = lifetime;
1641 }
1642}
1643
1644SharedPointer<NetworkSyscalls> NetworkSyscalls::acquireDescriptorLifetime() const {
1645 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1646 return m_DescriptorLifetime;
1647}
1648
1649SharedPointer<NetworkSyscalls> NetworkSyscalls::releaseDescriptorLifetime() {
1651 {
1652 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1653 lifetime = pedigree_std::move(m_DescriptorLifetime);
1654 }
1655 return lifetime;
1656}
1657
1659 if (!beginDescriptorClose()) {
1660 return;
1661 }
1662
1664 closeReadiness(ReadyInvalid | ReadyHangup);
1665}
1666
1667bool NetworkSyscalls::monitor(Thread* pThread, Event* pEvent) {
1668 return false;
1669}
1670
1671bool NetworkSyscalls::unmonitor(Event* pEvent) {
1672 return false;
1673}
1674
1675void NetworkSyscalls::associate(FileDescriptor* fd) {
1676 m_Blocking = !fd || !(fd->getStatusFlags() & O_NONBLOCK);
1677}
1678
1679bool NetworkSyscalls::isBlocking() const {
1680 return m_Blocking;
1681}
1682
1683void NetworkSyscalls::setBlocking(bool blocking) {
1684 m_Blocking = blocking;
1685}
1686
1688 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1689 if (m_LastDescriptorClosed) {
1690 return false;
1691 }
1692
1693 m_DescriptorAdmissionOpen = false;
1694 m_LastDescriptorClosed = true;
1695 return true;
1696}
1697
1698bool NetworkSyscalls::hasLastDescriptorClosed() const {
1699 ConstexprLockGuard<Mutex, THREADS> guard(m_LifecycleLock);
1700 return m_LastDescriptorClosed;
1701}
1702
1703LwipSocketSyscalls::LwipSocketSyscalls(int domain, int type, int protocol)
1704 : NetworkSyscalls(domain, type, protocol), m_Socket(nullptr), m_ReceiveLock(), m_Metadata() {}
1705
1706LwipSocketSyscalls::~LwipSocketSyscalls() {
1708}
1709
1711 if (!beginDescriptorClose()) {
1712 return;
1713 }
1714
1715 struct netconn* socket = m_Socket;
1716 if (socket) {
1717 LOCK_TCPIP_CORE();
1718 {
1719 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
1720 m_SyscallObjects.remove(socket);
1721 }
1722 UNLOCK_TCPIP_CORE();
1723 }
1724
1725 // A callback which acquired admission before the map removal may still be
1726 // finishing its metadata update. Drain it before publishing terminal state
1727 // or releasing the netconn.
1729
1730 struct pbuf* partialPacket = nullptr;
1731 struct netbuf* partialBuffer = nullptr;
1732 {
1733 ConstexprLockGuard<Mutex, THREADS> receiveGuard(m_ReceiveLock);
1734 {
1735 ConstexprLockGuard<Mutex, THREADS> metadataGuard(m_Metadata.lock);
1736 const ReadyMask previous = readinessLevelLocked();
1737 m_Metadata.closed = true;
1738 m_Metadata.peerClosed = true;
1739 m_Metadata.writeClosed = true;
1740 partialPacket = m_Metadata.pb;
1741 partialBuffer = m_Metadata.buf;
1742 m_Metadata.pb = nullptr;
1743 m_Metadata.buf = nullptr;
1744 m_Metadata.offset = 0;
1745 m_Metadata.partialRead = false;
1746 m_Metadata.receivingQueuedData = false;
1747 recordReadinessRisesLocked(previous);
1748 }
1749 }
1750
1751 m_Socket = nullptr;
1752 if (partialBuffer) {
1753 netbuf_delete(partialBuffer);
1754 } else if (partialPacket) {
1755 pbuf_free(partialPacket);
1756 }
1757 if (socket) {
1758 netconn_delete(socket);
1759 }
1760
1761 closeReadiness(ReadyInvalid | ReadyHangup);
1762}
1763
1764void LwipSocketSyscalls::setBlocking(bool blocking) {
1765 NetworkSyscalls::setBlocking(blocking);
1766 if (m_Socket) {
1767 netconn_set_nonblocking(m_Socket, blocking ? 0 : 1);
1768 }
1769}
1770
1771void LwipSocketSyscalls::registerSocket() {
1772 LOCK_TCPIP_CORE();
1773 {
1774 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
1775 if (!m_SyscallObjects.lookup(m_Socket)) {
1776 // lwIP counts receive events in socket while an accepted
1777 // connection has no userspace descriptor. Transfer those events
1778 // before exposing it so early request data remains readable.
1779 if (m_Socket->socket < 0) {
1780 ConstexprLockGuard<Mutex, THREADS> metadataGuard(m_Metadata.lock);
1781 const ReadyMask previous = readinessLevelLocked();
1782 m_Metadata.recv += -1 - m_Socket->socket;
1783 m_Socket->socket = 0;
1784 recordReadinessRisesLocked(previous);
1785 }
1786 m_SyscallObjects.insert(m_Socket, this);
1787 }
1788 }
1789 UNLOCK_TCPIP_CORE();
1790}
1791
1793 netconn_type connType = NETCONN_INVALID;
1794
1795 // fix up some defaults that make sense for inet[6] sockets
1796 if (!m_Protocol) {
1797 N_NOTICE("LwipSocketSyscalls: using default protocol for socket type");
1798 if (m_Type == SOCK_DGRAM) {
1799 m_Protocol = IPPROTO_UDP;
1800 } else if (m_Type == SOCK_STREAM) {
1801 m_Protocol = IPPROTO_TCP;
1802 }
1803 }
1804
1805 if (m_Domain == AF_INET) {
1806 switch (m_Protocol) {
1807 case IPPROTO_TCP:
1808 connType = NETCONN_TCP;
1809 break;
1810 case IPPROTO_UDP:
1811 connType = NETCONN_UDP;
1812 break;
1813 }
1814 } else if (m_Domain == AF_INET6) {
1815 switch (m_Protocol) {
1816 case IPPROTO_TCP:
1817 connType = NETCONN_TCP_IPV6;
1818 break;
1819 case IPPROTO_UDP:
1820 connType = NETCONN_UDP_IPV6;
1821 break;
1822 }
1823 } else if (m_Domain == AF_PACKET) {
1824 connType = NETCONN_RAW;
1825 } else {
1826 WARNING("LwipSocketSyscalls: domain " << m_Domain << " is not known!");
1827 SYSCALL_ERROR(InvalidArgument);
1828 return false;
1829 }
1830
1831 if (connType == NETCONN_INVALID) {
1832 N_NOTICE("LwipSocketSyscalls: invalid socket creation parameters");
1833 SYSCALL_ERROR(InvalidArgument);
1834 return false;
1835 }
1836
1837 // Socket already exists? No need to do the rest.
1838 if (m_Socket) {
1839 registerSocket();
1840 return true;
1841 }
1842
1843 m_Socket = netconn_new_with_callback(connType, netconnCallback);
1844 if (!m_Socket) {
1846 return false;
1847 }
1848
1849 if (NETCONNTYPE_GROUP(m_Socket->type) != NETCONN_TCP) {
1850 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
1851 const ReadyMask previous = readinessLevelLocked();
1852 m_Metadata.send = 1;
1853 recordReadinessRisesLocked(previous);
1854 }
1855
1856 registerSocket();
1857
1858 return true;
1859}
1860
1861int LwipSocketSyscalls::connect(const struct sockaddr_storage* address, socklen_t addrlen) {
1862 ip_addr_t ipaddr;
1863 ByteSet(&ipaddr, 0, sizeof(ipaddr));
1864 uint16_t port = 0;
1865 err_t err = sockaddrToIpaddr(address, port, &ipaddr, false);
1866 if (err != ERR_OK) {
1867 N_NOTICE("failed to convert sockaddr");
1868 lwipToSyscallError(err);
1869 return -1;
1870 }
1871
1872 // set blocking status if needed
1873 bool blocking = isBlocking();
1874 netconn_set_nonblocking(m_Socket, blocking ? 0 : 1);
1875
1876 N_NOTICE("using socket " << m_Socket << "!");
1877 N_NOTICE(" -> connecting to remote " << ipaddr_ntoa(&ipaddr) << " on port " << Dec << port);
1878
1879 err = netconn_connect(m_Socket, &ipaddr, port);
1880 if (err != ERR_OK) {
1881 N_NOTICE(" -> lwip error");
1882 lwipToSyscallError(err);
1883 return -1;
1884 }
1885
1886 // need to allow writing immediately for non-tcp sockets
1888 if (NETCONNTYPE_GROUP(m_Socket->type) != NETCONN_TCP) {
1889 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
1890 const ReadyMask previous = readinessLevelLocked();
1891 m_Metadata.send = 1;
1892 recordReadinessRisesLocked(previous);
1893 }
1894
1895 N_NOTICE(" -> ok!");
1896 return 0;
1897}
1898
1899ssize_t LwipSocketSyscalls::sendto_msg(const struct msghdr* msghdr,
1900 const SharedPointer<SocketRights>& rights) {
1901 if (rights) {
1902 SYSCALL_ERROR(OperationNotSupported);
1903 return -1;
1904 }
1905
1906 err_t err;
1907 const bool tcp = NETCONNTYPE_GROUP(m_Socket->type) == NETCONN_TCP;
1908 ip_addr_t destination = {};
1909 uint16_t destinationPort = 0;
1910 bool hasDestination = false;
1911
1912 if (msghdr->msg_name) {
1913 // Preserve the existing connected-stream behavior while enabling the
1914 // destination-bearing datagram path.
1915 if (tcp) {
1916 SYSCALL_ERROR(Unimplemented);
1917 return -1;
1918 }
1919 if (m_Domain != AF_INET) {
1920 SYSCALL_ERROR(OperationNotSupported);
1921 return -1;
1922 }
1923 if (msghdr->msg_namelen < sizeof(struct sockaddr_in)) {
1924 SYSCALL_ERROR(InvalidArgument);
1925 return -1;
1926 }
1927
1928 const struct sockaddr_storage* address =
1929 reinterpret_cast<const struct sockaddr_storage*>(msghdr->msg_name);
1930 err = sockaddrToIpaddr(address, destinationPort, &destination, false);
1931 if (err != ERR_OK) {
1932 lwipToSyscallError(err);
1933 return -1;
1934 }
1935 hasDestination = true;
1936 }
1937
1938 if (tcp) {
1939 bool hasPayload = false;
1940 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
1941 if (msghdr->msg_iov[i].iov_len) {
1942 hasPayload = true;
1943 break;
1944 }
1945 }
1946 if (!hasPayload) {
1947 return 0;
1948 }
1949 }
1950
1951 // Can we send without blocking?
1952 bool sendAvailable = false;
1953 {
1954 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
1955 sendAvailable = m_Metadata.send != 0;
1956 }
1957 if (!isBlocking() && !sendAvailable) {
1958 N_NOTICE(" -> send queue full, would block");
1959 SYSCALL_ERROR(NoMoreProcesses);
1960 return -1;
1961 }
1962
1963 size_t bytesWritten = 0;
1964 bool ok = true;
1965
1966 if (tcp) {
1967 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
1968 void* buffer = msghdr->msg_iov[i].iov_base;
1969 size_t bufferlen = msghdr->msg_iov[i].iov_len;
1970 if (!bufferlen) {
1971 continue;
1972 }
1973
1974 size_t thisBytesWritten = 0;
1975 err = netconn_write_partly(m_Socket, buffer, bufferlen, NETCONN_COPY | NETCONN_MORE,
1976 &thisBytesWritten);
1977 if (err != ERR_OK) {
1978 lwipToSyscallError(err);
1979 ok = false;
1980 break;
1981 }
1982
1983 bytesWritten += thisBytesWritten;
1984 if (thisBytesWritten < bufferlen) {
1985 break;
1986 }
1987 }
1988 } else {
1989 NetbufOwner buffer = NetbufOwner::adopt(netbuf_new());
1990 if (!buffer) {
1991 SYSCALL_ERROR(OutOfMemory);
1992 return -1;
1993 }
1994
1995 size_t datagramLength = 0;
1996 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
1997 const size_t fragmentLength = msghdr->msg_iov[i].iov_len;
1998 if (fragmentLength > static_cast<size_t>(0xFFFF) - datagramLength) {
1999 SYSCALL_ERROR(TooBig);
2000 return -1;
2001 }
2002 datagramLength += fragmentLength;
2003 }
2004
2005 char* payload =
2006 reinterpret_cast<char*>(netbuf_alloc(buffer.get(), static_cast<u16_t>(datagramLength)));
2007 if (!payload) {
2008 SYSCALL_ERROR(OutOfMemory);
2009 return -1;
2010 }
2011
2012 size_t offset = 0;
2013 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
2014 const size_t fragmentLength = msghdr->msg_iov[i].iov_len;
2015 if (fragmentLength) {
2016 MemoryCopy(payload + offset, msghdr->msg_iov[i].iov_base, fragmentLength);
2017 offset += fragmentLength;
2018 }
2019 }
2020
2021 err = hasDestination ? netconn_sendto(m_Socket, buffer.get(), &destination, destinationPort)
2022 : netconn_send(m_Socket, buffer.get());
2023 if (err != ERR_OK) {
2024 lwipToSyscallError(err);
2025 ok = false;
2026 } else {
2027 // lwIP can prepend protocol headers to the netbuf while sending.
2028 bytesWritten += datagramLength;
2029 }
2030 }
2031
2032 if (!bytesWritten) {
2033 if (!ok) {
2034 return -1;
2035 }
2036 } else {
2037 // A later vector failure cannot replace bytes already sent with an
2038 // error at the syscall boundary or make that progress restartable.
2039 syscallError(0);
2040 }
2041
2042 return bytesWritten;
2043}
2044
2045ssize_t LwipSocketSyscalls::recvfrom_msg(struct msghdr* msghdr,
2047 if (rights) {
2048 rights->reset();
2049 }
2050
2051 // A duplicated descriptor shares the receive cursor and retained packet.
2052 // Serialize the whole receive operation, but never hold the metadata lock
2053 // across lwIP calls because its callback takes that lock.
2054 ConstexprLockGuard<Mutex, THREADS> receiveGuard(m_ReceiveLock);
2055
2056 const bool tcp = NETCONNTYPE_GROUP(netconn_type(m_Socket)) == NETCONN_TCP;
2057 const int inputFlags = msghdr->msg_flags;
2058 const bool blocking = isBlocking() && !(inputFlags & MSG_DONTWAIT);
2059 if ((inputFlags & MSG_TRUNC) && (tcp || m_Type != SOCK_DGRAM)) {
2060 SYSCALL_ERROR(OperationNotSupported);
2061 return -1;
2062 }
2063 if (msghdr->msg_name && tcp) {
2064 // Source address reporting for streams is unchanged by the UDP slice.
2065 SYSCALL_ERROR(Unimplemented);
2066 return -1;
2067 }
2068 if (msghdr->msg_name && m_Domain != AF_INET) {
2069 SYSCALL_ERROR(OperationNotSupported);
2070 return -1;
2071 }
2072 msghdr->msg_flags = 0;
2073 if (!msghdr->msg_name) {
2074 msghdr->msg_namelen = 0;
2075 }
2076
2077 size_t vectorIndex = 0;
2078 size_t vectorOffset = 0;
2079 while (vectorIndex < static_cast<size_t>(msghdr->msg_iovlen) &&
2080 !msghdr->msg_iov[vectorIndex].iov_len) {
2081 ++vectorIndex;
2082 }
2083 if (tcp && vectorIndex == static_cast<size_t>(msghdr->msg_iovlen)) {
2084 return 0;
2085 }
2086
2087 // No data to read right now.
2088 if (!blocking) {
2089 bool noData = false;
2090 {
2091 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2092 if (m_Metadata.closed) {
2093 return 0;
2094 }
2095 noData = !(m_Metadata.recv || m_Metadata.partialRead);
2096 }
2097
2098 if (noData) {
2099 // If an app tightly calls recv() and keeps hitting here, it'll
2100 // burn a lot of cycles for no good reason. Instead, reschedule to
2101 // reduce that tight spin.
2103
2104 N_NOTICE(" -> no more data available, would block");
2105 SYSCALL_ERROR(NoMoreProcesses);
2106 return -1;
2107 }
2108 }
2109
2110 size_t totalLen = 0;
2111 size_t packetLength = 0;
2112 do {
2113 if (!m_Metadata.pb) {
2114 struct pbuf* pb = nullptr;
2115 struct netbuf* buf = nullptr;
2116
2117 {
2118 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2119 // The receive lock excludes other consumers. Once bytes have been
2120 // copied, only dequeue packets already announced by lwIP: a stream
2121 // read must not wait for the caller's entire buffer to fill.
2122 if (totalLen && !m_Metadata.recv) {
2123 break;
2124 }
2125 // RCV- is delivered synchronously from netconn_recv. Preserve the
2126 // already-readable level until the dequeued packet is installed as the
2127 // partial buffer, so epoll cannot observe an internal handoff as a drain.
2128 m_Metadata.receivingQueuedData = m_Metadata.recv != 0;
2129 }
2130
2131 // No partial data present from a previous read. Read new data from
2132 // the socket.
2133 err_t err;
2134 if (tcp) {
2135 err = netconn_recv_tcp_pbuf(m_Socket, &pb);
2136 } else {
2137 err = netconn_recv(m_Socket, &buf);
2138 }
2139
2140 if (err != ERR_OK) {
2141 if (err == ERR_CLSD) {
2142 ReadyMask changed = ReadyRead | ReadyReadHangup;
2143 {
2144 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2145 const ReadyMask previous = readinessLevelLocked();
2146 m_Metadata.receivingQueuedData = false;
2147 m_Metadata.closed = true;
2148 m_Metadata.peerClosed = true;
2149 if (m_Metadata.writeClosed) {
2150 changed |= ReadyHangup;
2151 }
2152 recordReadinessRisesLocked(previous);
2153 }
2154 notifyReadiness(changed);
2155 break;
2156 }
2157
2158 {
2159 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2160 m_Metadata.receivingQueuedData = false;
2161 }
2162 notifyReadiness(ReadyRead);
2163 N_NOTICE(" -> lwIP error");
2164 if (!totalLen) {
2165 lwipToSyscallError(err);
2166 return -1;
2167 }
2168 break;
2169 }
2170
2171 if (pb == nullptr && buf != nullptr) {
2172 pb = buf->p;
2173 }
2174 if (!pb) {
2175 {
2176 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2177 m_Metadata.receivingQueuedData = false;
2178 }
2179 notifyReadiness(ReadyRead);
2180 if (!totalLen) {
2181 SYSCALL_ERROR(IoError);
2182 return -1;
2183 }
2184 break;
2185 }
2186
2187 {
2188 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2189 m_Metadata.offset = 0;
2190 m_Metadata.pb = pb;
2191 m_Metadata.buf = buf;
2192 m_Metadata.partialRead = true;
2193 m_Metadata.receivingQueuedData = false;
2194 }
2195 }
2196
2197 packetLength = m_Metadata.pb->tot_len;
2198 if (!tcp && msghdr->msg_name) {
2199 const ip_addr_t* sourceAddress = netbuf_fromaddr(m_Metadata.buf);
2200 const uint16_t sourcePort = netbuf_fromport(m_Metadata.buf);
2201 struct sockaddr_in source = {};
2202 source.sin_family = AF_INET;
2203 source.sin_port = HOST_TO_BIG16(sourcePort);
2204 source.sin_addr.s_addr = ip_addr_get_ip4_u32(sourceAddress);
2205
2206 const size_t addressCapacity = msghdr->msg_namelen;
2207 const size_t addressLength =
2208 addressCapacity < sizeof(source) ? addressCapacity : sizeof(source);
2209 if (addressLength) {
2210 MemoryCopy(msghdr->msg_name, &source, addressLength);
2211 }
2212 msghdr->msg_namelen = sizeof(source);
2213 }
2214
2215 size_t readOffset = m_Metadata.offset;
2216 while (vectorIndex < static_cast<size_t>(msghdr->msg_iovlen) && readOffset < packetLength) {
2217 const struct iovec& vector = msghdr->msg_iov[vectorIndex];
2218 size_t bufferlen = vector.iov_len - vectorOffset;
2219 const size_t available = packetLength - readOffset;
2220 if (bufferlen > available) {
2221 bufferlen = available;
2222 }
2223 if (bufferlen) {
2224 pbuf_copy_partial(m_Metadata.pb, reinterpret_cast<uint8_t*>(vector.iov_base) + vectorOffset,
2225 bufferlen, readOffset);
2226 totalLen += bufferlen;
2227 readOffset += bufferlen;
2228 vectorOffset += bufferlen;
2229 }
2230 if (vectorOffset == vector.iov_len) {
2231 ++vectorIndex;
2232 vectorOffset = 0;
2233 }
2234 }
2235
2236 while (vectorIndex < static_cast<size_t>(msghdr->msg_iovlen) &&
2237 !msghdr->msg_iov[vectorIndex].iov_len) {
2238 ++vectorIndex;
2239 }
2240
2241 // TCP retains unread bytes as a stream cursor. Datagram reads consume one
2242 // whole packet and report that the caller's scatter buffer was too short.
2243 if (tcp && readOffset < packetLength) {
2244 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2245 m_Metadata.offset = readOffset;
2246 } else {
2247 struct pbuf* completedPacket = nullptr;
2248 struct netbuf* completedBuffer = nullptr;
2249 {
2250 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2251 completedPacket = m_Metadata.pb;
2252 completedBuffer = m_Metadata.buf;
2253 m_Metadata.pb = nullptr;
2254 m_Metadata.buf = nullptr;
2255 m_Metadata.offset = 0;
2256 m_Metadata.partialRead = false;
2257 }
2258
2259 if (!tcp && readOffset < packetLength) {
2260 msghdr->msg_flags |= MSG_TRUNC;
2261 }
2262
2263 if (completedBuffer) {
2264 netbuf_delete(completedBuffer);
2265 } else if (completedPacket) {
2266 pbuf_free(completedPacket);
2267 }
2268 }
2269
2270 if (!tcp) {
2271 break;
2272 }
2273 } while (vectorIndex < static_cast<size_t>(msghdr->msg_iovlen));
2274
2275 // Publish both sides of the readable predicate. Edge-triggered epoll must
2276 // observe a fully drained packet before a later arrival can raise another
2277 // edge; partial packets remain readable when the observer rechecks.
2278 notifyReadiness(ReadyRead);
2279
2280 if (totalLen) {
2281 syscallError(0);
2282 }
2283 N_NOTICE(" -> " << totalLen);
2284 if (!tcp && (inputFlags & MSG_TRUNC) && totalLen < packetLength) {
2285 return packetLength;
2286 }
2287 return totalLen;
2288}
2289
2290int LwipSocketSyscalls::listen(int backlog) {
2291 err_t err = netconn_listen_with_backlog(m_Socket, backlog);
2292 if (err != ERR_OK) {
2293 N_NOTICE(" -> lwIP error");
2294 lwipToSyscallError(err);
2295 return -1;
2296 }
2297
2298 {
2299 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2300 m_Metadata.listening = true;
2301 }
2302
2303 return 0;
2304}
2305
2306int LwipSocketSyscalls::bind(const struct sockaddr_storage* address, socklen_t addrlen) {
2307 uint16_t port = 0;
2308 ip_addr_t ipaddr;
2309 err_t conversion = sockaddrToIpaddr(address, port, &ipaddr);
2310 if (conversion != ERR_OK) {
2311 lwipToSyscallError(conversion);
2312 return -1;
2313 }
2314
2315 err_t err = netconn_bind(m_Socket, &ipaddr, port);
2316 if (err != ERR_OK) {
2317 N_NOTICE(" -> lwIP error");
2318 lwipToSyscallError(err);
2319 return -1;
2320 }
2321
2322 return 0;
2323}
2324
2325int LwipSocketSyscalls::accept(struct sockaddr_storage* address, socklen_t* addrlen, int flags,
2326 DescriptorLease* accepted) {
2327 struct netconn* new_conn;
2328 err_t err = netconn_accept(m_Socket, &new_conn);
2329 if (err != ERR_OK) {
2330 N_NOTICE(" -> lwIP error");
2331 lwipToSyscallError(err);
2332 return -1;
2333 }
2334
2335 // get the new peer
2336 ip_addr_t peer;
2337 uint16_t port;
2338 err = netconn_peer(new_conn, &peer, &port);
2339 if (err != ERR_OK) {
2340 netconn_delete(new_conn);
2341 lwipToSyscallError(err);
2342 return -1;
2343 }
2344
2346 struct sockaddr_in* sin = reinterpret_cast<struct sockaddr_in*>(address);
2347 sin->sin_family = AF_INET;
2348 sin->sin_port = HOST_TO_BIG16(port);
2349 sin->sin_addr.s_addr = peer.u_addr.ip4.addr;
2350 *addrlen = sizeof(sockaddr_in);
2351
2352 LwipSocketSyscalls* obj = new LwipSocketSyscalls(m_Domain, m_Type, m_Protocol);
2353 obj->m_Socket = new_conn;
2354 {
2355 ConstexprLockGuard<Mutex, THREADS> guard(obj->m_Metadata.lock);
2356 const ReadyMask previous = obj->readinessLevelLocked();
2357 obj->m_Metadata.send = 1;
2358 obj->recordReadinessRisesLocked(previous);
2359 }
2360 obj->create();
2361
2362 FileDescriptor* desc = new FileDescriptor;
2364 setSocketDescriptorFlags(desc, flags);
2365
2366 DescriptorLease installed;
2367
2368 const size_t fd = installDescriptor(desc, installed);
2369
2370 if (accepted) {
2371 *accepted = pedigree_std::move(installed);
2372 }
2373 obj->associate(desc);
2374
2375 return static_cast<int>(fd);
2376}
2377
2378int LwipSocketSyscalls::shutdown(int how) {
2379 int rx = 0;
2380 int tx = 0;
2381 if (how == SHUT_RDWR) {
2382 rx = tx = 1;
2383 } else if (how == SHUT_RD) {
2384 rx = 1;
2385 } else if (how == SHUT_WR) {
2386 tx = 1;
2387 } else {
2388 SYSCALL_ERROR(InvalidArgument);
2389 return -1;
2390 }
2391
2392 err_t err = netconn_shutdown(m_Socket, rx, tx);
2393 if (err != ERR_OK) {
2394 lwipToSyscallError(err);
2395 return -1;
2396 }
2397
2398 ReadyMask changed = ReadyNone;
2399 {
2400 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2401 const ReadyMask previous = readinessLevelLocked();
2402 if (rx) {
2403 m_Metadata.closed = true;
2404 changed |= ReadyRead | ReadyReadHangup;
2405 }
2406 if (tx) {
2407 m_Metadata.writeClosed = true;
2408 changed |= ReadyWrite;
2409 }
2410 if ((m_Metadata.closed || m_Metadata.peerClosed) && m_Metadata.writeClosed) {
2411 changed |= ReadyHangup;
2412 }
2413 recordReadinessRisesLocked(previous);
2414 }
2415 notifyReadiness(changed);
2416
2417 return 0;
2418}
2419
2420int LwipSocketSyscalls::getpeername(struct sockaddr_storage* address, socklen_t* address_len) {
2421 ip_addr_t peer;
2422 uint16_t port;
2423 err_t err = netconn_peer(m_Socket, &peer, &port);
2424 if (err != ERR_OK) {
2425 N_NOTICE(" -> getpeername failed");
2426 lwipToSyscallError(err);
2427 return -1;
2428 }
2429
2431 struct sockaddr_in* sin = reinterpret_cast<struct sockaddr_in*>(address);
2432 sin->sin_family = AF_INET;
2433 sin->sin_port = HOST_TO_BIG16(port);
2434 sin->sin_addr.s_addr = peer.u_addr.ip4.addr;
2435 *address_len = sizeof(sockaddr_in);
2436
2437 return 0;
2438}
2439
2440int LwipSocketSyscalls::getsockname(struct sockaddr_storage* address, socklen_t* address_len) {
2441 ip_addr_t self;
2442 uint16_t port;
2443 err_t err = netconn_addr(m_Socket, &self, &port);
2444 if (err != ERR_OK) {
2445 lwipToSyscallError(err);
2446 return -1;
2447 }
2448
2450 struct sockaddr_in* sin = reinterpret_cast<struct sockaddr_in*>(address);
2451 sin->sin_family = AF_INET;
2452 sin->sin_port = HOST_TO_BIG16(port);
2453 sin->sin_addr.s_addr = self.u_addr.ip4.addr;
2454 *address_len = sizeof(sockaddr_in);
2455
2456 return 0;
2457}
2458
2459int LwipSocketSyscalls::setsockopt(int level, int optname, const void* optvalue, socklen_t optlen) {
2460 if (optlen < sizeof(int)) {
2461 SYSCALL_ERROR(InvalidArgument);
2462 return -1;
2463 }
2464
2465 const int value = *reinterpret_cast<const int*>(optvalue);
2466 if (level == SOL_SOCKET) {
2467 const uint8_t option = lwipSocketOption(optname);
2468 if (option) {
2469 LOCK_TCPIP_CORE();
2470 struct ip_pcb* pcb = m_Socket ? m_Socket->pcb.ip : nullptr;
2471 if (!pcb) {
2472 UNLOCK_TCPIP_CORE();
2473 SYSCALL_ERROR(InvalidArgument);
2474 return -1;
2475 }
2476
2477 if (value) {
2478 ip_set_option(pcb, option);
2479 } else {
2480 ip_reset_option(pcb, option);
2481 }
2482 UNLOCK_TCPIP_CORE();
2483 return 0;
2484 }
2485 }
2486
2487 if (m_Protocol == IPPROTO_TCP && level == IPPROTO_TCP) {
2488#pragma GCC diagnostic push
2489#pragma GCC diagnostic ignored "-Wold-style-cast"
2490 if (optname == linuxTcpNoDelay) {
2491 LOCK_TCPIP_CORE();
2492 struct tcp_pcb* pcb = m_Socket ? m_Socket->pcb.tcp : nullptr;
2493 if (!pcb) {
2494 UNLOCK_TCPIP_CORE();
2495 SYSCALL_ERROR(InvalidArgument);
2496 return -1;
2497 }
2498
2499 N_NOTICE(" -> TCP_NODELAY");
2500 N_NOTICE(" --> val=" << value);
2501
2502 // TCP_NODELAY controls Nagle's algorithm usage
2503 if (value) {
2504 tcp_nagle_disable(pcb);
2505 } else {
2506 tcp_nagle_enable(pcb);
2507 }
2508
2509 UNLOCK_TCPIP_CORE();
2510 return 0;
2511 }
2512#pragma GCC diagnostic pop
2513 }
2514
2515 SYSCALL_ERROR(ProtocolNotAvailable);
2516 return -1;
2517}
2518
2519int LwipSocketSyscalls::getsockopt(int level, int optname, void* optvalue, socklen_t* optlen) {
2520 if (*optlen < sizeof(int)) {
2521 SYSCALL_ERROR(InvalidArgument);
2522 return -1;
2523 }
2524
2525 int value = 0;
2526 bool clearedError = false;
2527 if (level == SOL_SOCKET) {
2528 if (optname == SO_TYPE) {
2529 value = m_Type;
2530 } else if (optname == SO_ERROR) {
2531 {
2532 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2533 const ReadyMask previous = readinessLevelLocked();
2534 value = lwipErrorNumber(m_Metadata.error);
2535 clearedError = m_Metadata.error != ERR_OK;
2536 m_Metadata.error = ERR_OK;
2537 recordReadinessRisesLocked(previous);
2538 }
2539 } else {
2540 const uint8_t option = lwipSocketOption(optname);
2541 if (!option) {
2542 SYSCALL_ERROR(ProtocolNotAvailable);
2543 return -1;
2544 }
2545 LOCK_TCPIP_CORE();
2546 struct ip_pcb* pcb = m_Socket ? m_Socket->pcb.ip : nullptr;
2547 if (!pcb) {
2548 UNLOCK_TCPIP_CORE();
2549 SYSCALL_ERROR(InvalidArgument);
2550 return -1;
2551 }
2552 value = ip_get_option(pcb, option) ? 1 : 0;
2553 UNLOCK_TCPIP_CORE();
2554 }
2555 } else if (m_Protocol == IPPROTO_TCP && level == IPPROTO_TCP && optname == linuxTcpNoDelay) {
2556 LOCK_TCPIP_CORE();
2557 struct tcp_pcb* pcb = m_Socket ? m_Socket->pcb.tcp : nullptr;
2558 if (!pcb) {
2559 UNLOCK_TCPIP_CORE();
2560 SYSCALL_ERROR(InvalidArgument);
2561 return -1;
2562 }
2563 value = tcp_nagle_disabled(pcb) ? 1 : 0;
2564 UNLOCK_TCPIP_CORE();
2565 } else {
2566 SYSCALL_ERROR(ProtocolNotAvailable);
2567 return -1;
2568 }
2569
2570 *reinterpret_cast<int*>(optvalue) = value;
2571 *optlen = sizeof(int);
2572 if (clearedError) {
2573 notifyReadiness(ReadyError | ReadyWrite);
2574 }
2575 return 0;
2576}
2577
2578bool LwipSocketSyscalls::canPoll() const {
2579 return true;
2580}
2581
2582ReadyMask LwipSocketSyscalls::readinessLevelLocked() const {
2583 ReadyMask ready = ReadyNone;
2584 if (m_Metadata.recv || m_Metadata.partialRead || m_Metadata.receivingQueuedData ||
2585 m_Metadata.closed || m_Metadata.peerClosed) {
2586 ready |= ReadyRead;
2587 }
2588 if (m_Metadata.send && m_Metadata.error == ERR_OK) {
2589 ready |= ReadyWrite;
2590 }
2591 if (m_Metadata.closed || m_Metadata.peerClosed) {
2592 ready |= ReadyReadHangup;
2593 }
2594 if ((m_Metadata.closed || m_Metadata.peerClosed) && m_Metadata.writeClosed) {
2595 ready |= ReadyHangup;
2596 }
2597 if (m_Metadata.error != ERR_OK) {
2598 ready |= ReadyError;
2599 }
2600
2601 return ready;
2602}
2603
2604void LwipSocketSyscalls::recordReadinessRisesLocked(ReadyMask previous) {
2605 const ReadyMask current = readinessLevelLocked();
2606 if (!(previous & ReadyRead) && (current & ReadyRead)) {
2607 ++m_Metadata.generations.read;
2608 }
2609 if (!(previous & ReadyWrite) && (current & ReadyWrite)) {
2610 ++m_Metadata.generations.write;
2611 }
2612 if (!(previous & ReadyError) && (current & ReadyError)) {
2613 ++m_Metadata.generations.error;
2614 }
2615 if (!(previous & ReadyHangup) && (current & ReadyHangup)) {
2616 ++m_Metadata.generations.hangup;
2617 }
2618 if (!(previous & ReadyReadHangup) && (current & ReadyReadHangup)) {
2619 ++m_Metadata.generations.readHangup;
2620 }
2621}
2622
2623ReadyMask LwipSocketSyscalls::queryReady(bool reading, bool writing) {
2624 // An epoll watch can outlive the last descriptor alias. Admit the complete
2625 // snapshot before touching transport state so close waits for this query.
2628 return ReadyInvalid | ReadyHangup;
2629 }
2630
2631 if (hasLastDescriptorClosed()) {
2632 return ReadyInvalid | ReadyHangup;
2633 }
2634
2635 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2636 ReadyMask ready = readinessLevelLocked();
2637 if (!reading) {
2638 ready &= ~ReadyRead;
2639 }
2640 if (!writing) {
2641 ready &= ~ReadyWrite;
2642 }
2643 return ready | pendingReceiveReadiness();
2644}
2645
2648 if (!m_ReadinessNotifications.tryAcquire(query) || hasLastDescriptorClosed()) {
2649 return ReadinessGenerations();
2650 }
2651
2652 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2653 return withReceiveErrorGeneration(m_Metadata.generations);
2654}
2655
2656bool LwipSocketSyscalls::poll(bool& read, bool& write, bool& error, Semaphore* waiter) {
2657 bool ok = false;
2658
2659 if (!(read || write || error)) {
2660 // not actually polling for anything
2661 return true;
2662 }
2663
2664 ConstexprLockGuard<Mutex, THREADS> guard(m_Metadata.lock);
2665
2666 if (write) {
2667 write = m_Metadata.send != 0;
2668 ok = ok || write;
2669 }
2670
2671 if (read) {
2672 read = m_Metadata.recv || m_Metadata.partialRead || m_Metadata.closed || m_Metadata.peerClosed;
2673 ok = ok || read;
2674 }
2675
2676 if (error) {
2677 error = m_Metadata.error != ERR_OK || pendingReceiveReadiness();
2678 ok = ok || error;
2679 }
2680
2681 if (waiter && !ok) {
2682 // Need to wait for socket data.
2684 m_Metadata.semaphores.pushBack(waiter);
2685 }
2686
2687 return ok;
2688}
2689
2690void LwipSocketSyscalls::unPoll(Semaphore* waiter) {
2691 m_Metadata.lock.acquire();
2692 for (auto it = m_Metadata.semaphores.begin(); it != m_Metadata.semaphores.end();) {
2693 if ((*it) == waiter) {
2694 it = m_Metadata.semaphores.erase(it);
2695 } else {
2696 ++it;
2697 }
2698 }
2699 m_Metadata.lock.release();
2700}
2701
2702void LwipSocketSyscalls::netconnCallback(struct netconn* conn, enum netconn_evt evt, u16_t len) {
2703 LwipSocketSyscalls* obj = nullptr;
2704 OperationBarrier::Lease notification;
2705 {
2706 ConstexprLockGuard<Mutex, THREADS> objectsGuard(m_SyscallObjectsLock);
2707 obj = m_SyscallObjects.lookup(conn);
2708 if (!obj) {
2709 // Accepted netconns can receive data before accept() has associated a
2710 // Pedigree descriptor. lwIP initializes socket to -1 for this exact
2711 // handoff and invokes this callback while holding its core lock.
2712 if (conn && conn->socket < 0 && evt == NETCONN_EVT_RCVPLUS) {
2713 --conn->socket;
2714 }
2715 return;
2716 }
2717
2718 if (!obj->m_ReadinessNotifications.tryAcquire(notification)) {
2719 return;
2720 }
2721 }
2722
2723 ReadyMask changed = ReadyNone;
2724 {
2725 ConstexprLockGuard<Mutex, THREADS> guard(obj->m_Metadata.lock);
2726 const ReadyMask previous = obj->readinessLevelLocked();
2727
2728 switch (evt) {
2729 case NETCONN_EVT_RCVPLUS:
2730 N_NOTICE("RCV+");
2731 ++(obj->m_Metadata.recv);
2732 changed |= ReadyRead;
2733 if (NETCONNTYPE_GROUP(conn->type) == NETCONN_TCP && !len && !obj->m_Metadata.listening) {
2734 obj->m_Metadata.peerClosed = true;
2735 changed |= ReadyReadHangup;
2736 if (obj->m_Metadata.writeClosed) {
2737 changed |= ReadyHangup;
2738 }
2739 }
2740 break;
2741 case NETCONN_EVT_RCVMINUS:
2742 N_NOTICE("RCV-");
2743 if (obj->m_Metadata.recv) {
2744 --(obj->m_Metadata.recv);
2745 }
2746 break;
2747 case NETCONN_EVT_SENDPLUS:
2748 N_NOTICE("SND+");
2749 obj->m_Metadata.send = 1;
2750 changed |= ReadyWrite;
2751 break;
2752 case NETCONN_EVT_SENDMINUS:
2753 N_NOTICE("SND-");
2754 obj->m_Metadata.send = 0;
2755 changed |= ReadyWrite;
2756 break;
2757 case NETCONN_EVT_ERROR:
2758 N_NOTICE("ERR");
2759 obj->m_Metadata.error = netconn_err(conn);
2760 if (obj->m_Metadata.error == ERR_OK) {
2761 obj->m_Metadata.error = ERR_IF;
2762 }
2763 changed |= ReadyError;
2764 break;
2765 default:
2766 N_NOTICE("Unknown netconn callback error.");
2767 }
2768
2769 obj->recordReadinessRisesLocked(previous);
2770
2772 EMIT_IF(THREADS) {
2773 for (auto& it : obj->m_Metadata.semaphores) {
2774 it->release();
2775 }
2776 }
2777 }
2778
2779 // Observers can re-enter queryReady(), which takes m_Metadata.lock.
2780 obj->notifyReadiness(changed);
2781}
2782
2783void LwipSocketSyscalls::lwipToSyscallError(err_t err) {
2784 if (err != ERR_OK) {
2785 N_NOTICE(" -> lwip strerror gives '" << lwip_strerr(err) << "'");
2786 syscallError(lwipErrorNumber(err));
2787 }
2788}
2789
2790LwipSocketSyscalls::LwipMetadata::LwipMetadata()
2791 : recv(0),
2792 send(0),
2793 error(ERR_OK),
2794 closed(false),
2795 peerClosed(false),
2796 writeClosed(false),
2797 listening(false),
2798 partialRead(false),
2799 receivingQueuedData(false),
2800 lock(),
2801 semaphores(),
2802 offset(0),
2803 pb(nullptr),
2804 buf(nullptr),
2805 generations() {}
2806
2807enum class UnixSocketReferenceOwnership { Heap, Vfs };
2808
2810 public:
2811 UnixSocketReference(UnixSocket* socket, UnixSocketReferenceOwnership ownership)
2812 : m_Socket(socket), m_Ownership(ownership) {}
2813
2815 if (!m_Socket) {
2816 return;
2817 }
2818
2819 if (m_Ownership == UnixSocketReferenceOwnership::Vfs) {
2820 releaseTrackedUnixSocket(m_Socket);
2821 } else {
2822 delete m_Socket;
2823 }
2824 }
2825
2826 UnixSocket* get() const {
2827 return m_Socket;
2828 }
2829
2830 private:
2831 UnixSocket* m_Socket;
2832 UnixSocketReferenceOwnership m_Ownership;
2833};
2834
2835static HashTable<String, SharedPointer<UnixSocketReference>> g_AbstractUnixSockets;
2836static Mutex g_AbstractUnixSocketsLock;
2837
2839 public:
2841 : m_Reference(reference), m_Retired(false) {}
2842
2844 retire();
2845
2846 UnixSocket* socket = get();
2847 List<UnixSocket*> peers;
2848 UnixSocketSyscalls::unregisterSocket(socket, peers);
2849 for (auto peer : peers) {
2850 UnixSocketSyscalls::notifySocket(
2851 peer, ReadyRead | ReadyWrite | ReadyError | ReadyReadHangup | ReadyHangup);
2852 }
2853
2854 // The directory owns a bound pathname until unlink, independently of the
2855 // descriptor's endpoint lifetime. A closed endpoint remains unconnectable.
2856 }
2857
2858 UnixSocket* get() const {
2859 return m_Reference ? m_Reference->get() : nullptr;
2860 }
2861
2862 SharedPointer<UnixSocketReference> reference() const {
2863 return m_Reference;
2864 }
2865
2866 void retire() {
2867 if (m_Retired.compareAndSwap(false, true)) {
2868 UnixSocket* socket = get();
2869 if (socket) {
2870 socket->unbind();
2871 }
2872 }
2873 }
2874
2875 private:
2877 Atomic<bool> m_Retired;
2878};
2879
2880#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2881using UnixEndpointReceiveLeaseHook = void (*)();
2882static UnixEndpointReceiveLeaseHook g_UnixEndpointReceiveLeaseHook = nullptr;
2883static UnixEndpointMutationLockHook g_UnixEndpointMutationLockHook = nullptr;
2884static UnixEndpointReadinessLeaseHook g_UnixEndpointReadinessLeaseHook = nullptr;
2885
2886void invokeUnixEndpointMutationLockHook() {
2887 UnixEndpointMutationLockHook hook =
2888 __atomic_exchange_n(&g_UnixEndpointMutationLockHook,
2889 static_cast<UnixEndpointMutationLockHook>(nullptr), __ATOMIC_ACQ_REL);
2890 if (hook) {
2891 hook();
2892 }
2893}
2894
2895void invokeUnixEndpointReadinessLeaseHook() {
2896 UnixEndpointReadinessLeaseHook hook =
2897 __atomic_exchange_n(&g_UnixEndpointReadinessLeaseHook,
2898 static_cast<UnixEndpointReadinessLeaseHook>(nullptr), __ATOMIC_ACQ_REL);
2899 if (hook) {
2900 hook();
2901 }
2902}
2903#endif
2904
2905#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2906void setUnixEndpointMutationLockHookForTest(UnixEndpointMutationLockHook hook) {
2907 __atomic_store_n(&g_UnixEndpointMutationLockHook, hook, __ATOMIC_RELEASE);
2908}
2909
2910void setUnixEndpointReadinessLeaseHookForTest(UnixEndpointReadinessLeaseHook hook) {
2911 __atomic_store_n(&g_UnixEndpointReadinessLeaseHook, hook, __ATOMIC_RELEASE);
2912}
2913#endif
2914
2915UnixSocketSyscalls::EndpointMutationGuard::EndpointMutationGuard(UnixSocketSyscalls& socket)
2916 : m_Socket(socket), m_Guard(socket.m_EndpointMutationLock) {
2917#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2918 invokeUnixEndpointMutationLockHook();
2919#endif
2920}
2921
2922UnixSocketSyscalls::EndpointMutationGuard::~EndpointMutationGuard() {
2923 m_Socket.m_EndpointMutationReleaseInProgress = true;
2925 m_Guard.disown();
2926 m_Socket.m_EndpointMutationReleaseInProgress = false;
2927 m_Socket.tryCompleteEndpointClose();
2928}
2929
2930UnixSocketSyscalls::EndpointMutationPairGuard::EndpointMutationPairGuard(UnixSocketSyscalls& first,
2931 UnixSocketSyscalls& second)
2932 : m_First(first),
2933 m_Second(second),
2934 m_FirstGuard(first.m_EndpointMutationLock),
2935 m_SecondGuard(second.m_EndpointMutationLock) {
2936#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2937 invokeUnixEndpointMutationLockHook();
2938#endif
2939}
2940
2941UnixSocketSyscalls::EndpointMutationPairGuard::~EndpointMutationPairGuard() {
2942 m_First.m_EndpointMutationReleaseInProgress = true;
2943 m_Second.m_EndpointMutationReleaseInProgress = true;
2944
2946 m_SecondGuard.disown();
2948 m_FirstGuard.disown();
2949
2950 m_First.m_EndpointMutationReleaseInProgress = false;
2951 m_Second.m_EndpointMutationReleaseInProgress = false;
2952 m_First.tryCompleteEndpointClose();
2953 m_Second.tryCompleteEndpointClose();
2954}
2955
2956UnixSocketSyscalls::EndpointReadinessGuard::EndpointReadinessGuard()
2957 : m_Socket(nullptr), m_Lifetime(), m_Lease(), m_Acquired(false) {}
2958
2959UnixSocketSyscalls::EndpointReadinessGuard::EndpointReadinessGuard(UnixSocketSyscalls& socket)
2960 : EndpointReadinessGuard() {
2961 const bool acquired = acquire(socket);
2962 (void)acquired;
2963}
2964
2965bool UnixSocketSyscalls::EndpointReadinessGuard::acquire(UnixSocketSyscalls& socket) {
2966 assert(!m_Acquired);
2967 m_Socket = &socket;
2968 m_Lifetime = socket.acquireDescriptorLifetime();
2969 m_Acquired = socket.m_ReadinessNotifications.tryAcquire(m_Lease);
2970#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
2971 if (m_Acquired) {
2972 invokeUnixEndpointReadinessLeaseHook();
2973 }
2974#endif
2975 if (!m_Acquired) {
2976 m_Lifetime.reset();
2977 m_Socket = nullptr;
2978 }
2979 return m_Acquired;
2980}
2981
2982UnixSocketSyscalls::EndpointReadinessGuard::~EndpointReadinessGuard() {
2983 if (!m_Acquired) {
2984 return;
2985 }
2986
2987 m_Acquired = false;
2988 m_Lease = OperationBarrier::Lease();
2989 m_Socket->tryCompleteEndpointClose();
2990}
2991
2992UnixSocketSyscalls::UnixSocketSyscalls(int domain, int type, int protocol)
2993 : NetworkSyscalls(domain, type, protocol),
2996 m_EndpointMutationReleaseInProgress(false),
2997 m_EndpointClosePending(false),
2998 m_EndpointRetired(false),
2999 m_EndpointCloseFinalized(false),
3000 m_LocalEndpoint(),
3001 m_RemoteEndpoint(),
3002 m_ClosingLocalEndpoint(),
3003 m_ClosingRemoteEndpoint(),
3004 m_LocalPath(),
3005 m_RemotePath(),
3006 m_OwnsAbstractName(false) {}
3007
3008UnixSocketSyscalls::~UnixSocketSyscalls() {
3010}
3011
3012void UnixSocketSyscalls::registerSocket(UnixSocket* socket) {
3013 if (!socket) {
3014 return;
3015 }
3016
3017 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
3018 UnixSocketSyscalls* current = m_SyscallObjects.lookup(socket);
3019 if (!current) {
3020 m_SyscallObjects.insert(socket, this);
3021 } else if (current != this) {
3022 FATAL("A Unix socket has multiple NetworkSyscalls owners.");
3023 }
3024
3025 // Accepted sockets are registered after leaving the listener queue.
3026 m_PendingListeners.remove(socket);
3027}
3028
3029void UnixSocketSyscalls::registerPeer(UnixSocket* socket, UnixSocket* peer, UnixSocket* listener) {
3030 if (!socket || !peer) {
3031 return;
3032 }
3033
3034 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
3035 m_Peers.insert(socket, peer);
3036 m_Peers.insert(peer, socket);
3037 if (listener) {
3038 m_PendingListeners.insert(peer, listener);
3039 }
3040}
3041
3042void UnixSocketSyscalls::unregisterPeer(UnixSocket* socket, UnixSocket* peer) {
3043 if (!socket || !peer) {
3044 return;
3045 }
3046
3047 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
3048 if (m_Peers.lookup(socket) == peer) {
3049 m_Peers.remove(socket);
3050 }
3051 if (m_Peers.lookup(peer) == socket) {
3052 m_Peers.remove(peer);
3053 }
3054 m_PendingListeners.remove(peer);
3055}
3056
3057void UnixSocketSyscalls::unregisterSocket(UnixSocket* socket, List<UnixSocket*>& peers) {
3058 if (!socket) {
3059 return;
3060 }
3061
3062 List<UnixSocket*> pendingEndpoints;
3063 {
3064 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
3065 m_SyscallObjects.remove(socket);
3066
3067 UnixSocket* peer = m_Peers.lookup(socket);
3068 if (peer) {
3069 peers.pushBack(peer);
3070 m_Peers.remove(socket);
3071 if (m_Peers.lookup(peer) == socket) {
3072 m_Peers.remove(peer);
3073 }
3074 m_PendingListeners.remove(peer);
3075 }
3076 m_PendingListeners.remove(socket);
3077
3078 // A closing listener owns queued endpoints which do not have syscall
3079 // wrappers yet. Preserve their client endpoint keys until after unbind()
3080 // has changed the shared connection state, then notify those clients.
3081 for (Tree<UnixSocket*, UnixSocket*>::Iterator it = m_PendingListeners.begin();
3082 it != m_PendingListeners.end(); ++it) {
3083 if (it.value() == socket) {
3084 pendingEndpoints.pushBack(it.key());
3085 }
3086 }
3087
3088 for (auto pending : pendingEndpoints) {
3089 UnixSocket* pendingPeer = m_Peers.lookup(pending);
3090 if (pendingPeer) {
3091 peers.pushBack(pendingPeer);
3092 m_Peers.remove(pending);
3093 if (m_Peers.lookup(pendingPeer) == pending) {
3094 m_Peers.remove(pendingPeer);
3095 }
3096 }
3097 m_PendingListeners.remove(pending);
3098 }
3099 }
3100}
3101
3102void UnixSocketSyscalls::notifySocket(UnixSocket* socket, ReadyMask mask) {
3103 if (!socket || !mask) {
3104 return;
3105 }
3106
3107 UnixSocketSyscalls* target = nullptr;
3108 EndpointReadinessGuard notification;
3109 {
3110 ConstexprLockGuard<Mutex, THREADS> registryGuard(m_SyscallObjectsLock);
3111 target = m_SyscallObjects.lookup(socket);
3112 if (!target || !notification.acquire(*target)) {
3113 return;
3114 }
3115 }
3116
3117 // queryReady() may take UnixSocket's connection or buffer locks.
3118 target->notifyReadiness(mask);
3119}
3120
3121void UnixSocketSyscalls::notifyPeer(UnixSocket* socket, ReadyMask mask) {
3122 UnixSocket* peer = nullptr;
3123 {
3124 ConstexprLockGuard<Mutex, THREADS> guard(m_SyscallObjectsLock);
3125 if (socket) {
3126 peer = m_Peers.lookup(socket);
3127 }
3128 }
3129
3130 notifySocket(peer, mask);
3131}
3132
3133bool UnixSocketSyscalls::publishAbstractSocket(
3134 const String& address, const SharedPointer<UnixSocketReference>& reference) {
3135 const String key = abstractKey(address);
3136 LockGuard<Mutex> guard(g_AbstractUnixSocketsLock);
3137 if (g_AbstractUnixSockets.contains(key)) {
3138 SYSCALL_ERROR(AddressInUse);
3139 return false;
3140 }
3141 if (!g_AbstractUnixSockets.insert(key, reference)) {
3142 SYSCALL_ERROR(OutOfMemory);
3143 return false;
3144 }
3145 return true;
3146}
3147
3148SharedPointer<UnixSocketReference> UnixSocketSyscalls::acquireSocket(const String& address) {
3149 if (isAbstractUnixSocket(address)) {
3150 const String key = abstractKey(address);
3151 LockGuard<Mutex> guard(g_AbstractUnixSocketsLock);
3152 auto result = g_AbstractUnixSockets.lookup(key);
3153 if (!result.hasValue()) {
3154 SYSCALL_ERROR(DoesNotExist);
3156 }
3157 return result.value();
3158 }
3159
3160 File* file = findTrackedUnixSocket(address);
3161 if (!file) {
3162 SYSCALL_ERROR(DoesNotExist);
3164 }
3165 if (!file->isSocket()) {
3166 releaseTrackedUnixSocket(file);
3167 SYSCALL_ERROR(DoesNotExist);
3169 }
3171 new UnixSocketReference(static_cast<UnixSocket*>(file), UnixSocketReferenceOwnership::Vfs));
3172}
3173
3174void UnixSocketSyscalls::removeAbstractSocket(const String& address, UnixSocket* socket) {
3175 const String key = abstractKey(address);
3176 LockGuard<Mutex> guard(g_AbstractUnixSocketsLock);
3177 auto current = g_AbstractUnixSockets.lookup(key);
3178 if (current.hasValue() && current.value()->get() == socket) {
3179 g_AbstractUnixSockets.remove(key);
3180 }
3181}
3182
3183String UnixSocketSyscalls::abstractKey(const String& address) const {
3184 String key;
3185 // The transport discriminator is outside user-controlled address bytes.
3186 key.Format(
3187 "%llu:%d:",
3188 m_NetworkNamespace ? static_cast<unsigned long long>(m_NetworkNamespace->identity()) : 0ULL,
3189 m_Domain);
3190 key += address;
3191 return key;
3192}
3193
3194SharedPointer<UnixSocketGeneration> UnixSocketSyscalls::acquireLocalEndpoint() const {
3196 return m_LocalEndpoint;
3197}
3198
3199void UnixSocketSyscalls::replaceLocalEndpoint(UnixSocket* socket, bool tracked,
3200 const String* localPath) {
3202 socket, tracked ? UnixSocketReferenceOwnership::Vfs : UnixSocketReferenceOwnership::Heap));
3203 replaceLocalEndpoint(reference, localPath, false);
3204}
3205
3206void UnixSocketSyscalls::replaceLocalEndpoint(const SharedPointer<UnixSocketReference>& reference,
3207 const String* localPath, bool ownsAbstractName) {
3208 UnixSocket* socket = reference ? reference->get() : nullptr;
3211
3212 registerSocket(socket);
3213 {
3215 previous = pedigree_std::move(m_LocalEndpoint);
3216 m_LocalEndpoint = pedigree_std::move(replacement);
3217 if (localPath) {
3218 m_LocalPath = *localPath;
3219 }
3220 m_OwnsAbstractName = ownsAbstractName;
3221 }
3222
3223 if (previous) {
3224 previous->retire();
3225 }
3226}
3227
3229 const bool firstClose = beginDescriptorClose();
3230 if (firstClose) {
3231 // Admission and notification delivery close immediately. Endpoint
3232 // retirement may need to wait for a mutation scope or for the current
3233 // readiness callback/query to release its own activity lease.
3234 m_EndpointClosePending = true;
3236 } else if (!m_EndpointClosePending) {
3237 return;
3238 }
3239
3240#if THREADS
3242 return;
3243 }
3244#endif
3245 tryCompleteEndpointClose();
3246}
3247
3248void UnixSocketSyscalls::tryCompleteEndpointClose() {
3249 if (!m_EndpointClosePending || m_EndpointCloseFinalized || m_EndpointMutationReleaseInProgress) {
3250 return;
3251 }
3252
3253 // A published descriptor keeps one cycle solely for this deferred path.
3254 // The local copy prevents the last readiness lease from deleting this
3255 // wrapper while it completes the close after returning from a handler.
3256 SharedPointer<NetworkSyscalls> completionLifetime = acquireDescriptorLifetime();
3257
3258 if (!m_EndpointRetired) {
3259 TerminationDeferral terminationDeferral;
3261 return;
3262 }
3263
3264 // Another completion attempt can retire the endpoint after the first
3265 // unlocked check and before this nonblocking acquisition.
3266 if (!m_EndpointRetired) {
3267 m_EndpointMutationReleaseInProgress = true;
3268 String abstractName;
3269 UnixSocket* abstractSocket = nullptr;
3270 {
3272 if (m_OwnsAbstractName && isAbstractUnixSocket(m_LocalPath) && m_LocalEndpoint) {
3273 abstractName = m_LocalPath;
3274 abstractSocket = m_LocalEndpoint->get();
3275 }
3276 m_ClosingLocalEndpoint = pedigree_std::move(m_LocalEndpoint);
3277 m_ClosingRemoteEndpoint = pedigree_std::move(m_RemoteEndpoint);
3278 m_LocalPath.clear();
3279 m_RemotePath.clear();
3280 m_OwnsAbstractName = false;
3281 }
3282
3283 if (abstractSocket) {
3284 removeAbstractSocket(abstractName, abstractSocket);
3285 }
3286
3287 if (m_ClosingLocalEndpoint) {
3288 // Wake blocked reads and accepts. Their generation references keep the
3289 // retired object alive until those operations have observed closure.
3290 m_ClosingLocalEndpoint->retire();
3291 }
3292
3293 m_EndpointRetired = true;
3294 }
3296 m_EndpointMutationReleaseInProgress = false;
3297 }
3298
3299 if (!m_EndpointRetired || !m_ReadinessNotifications.isClosedAndDrained() ||
3300 !m_EndpointCloseFinalized.compareAndSwap(false, true)) {
3301 return;
3302 }
3303
3304 N_NOTICE("UnixSocketSyscalls::~UnixSocketSyscalls");
3305 m_ClosingLocalEndpoint.reset();
3306 m_ClosingRemoteEndpoint.reset();
3307 closeReadiness(ReadyInvalid | ReadyHangup);
3308 m_EndpointClosePending = false;
3309
3310 SharedPointer<NetworkSyscalls> publishedLifetime = releaseDescriptorLifetime();
3311}
3312
3314 EndpointMutationGuard mutationGuard(*this);
3315 if (hasLastDescriptorClosed()) {
3316 SYSCALL_ERROR(BadFileDescriptor);
3317 return false;
3318 }
3319
3320 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
3321 if (local) {
3322 registerSocket(local->get());
3323 return true;
3324 }
3325
3326 // Create an unnamed unix socket by default.
3327 replaceLocalEndpoint(
3328 new UnixSocket(String(), g_pUnixSocketBacking, nullptr, nullptr, getSocketType()), false);
3329
3330 return true;
3331}
3332
3333int UnixSocketSyscalls::connect(const struct sockaddr_storage* address, socklen_t addrlen) {
3334 String pathname;
3335 if (!unixSocketPath(address, addrlen, pathname, false)) {
3336 return -1;
3337 }
3338
3339 EndpointMutationGuard mutationGuard(*this);
3340 if (hasLastDescriptorClosed()) {
3341 SYSCALL_ERROR(BadFileDescriptor);
3342 return -1;
3343 }
3344 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
3345 if (!local) {
3346 SYSCALL_ERROR(BadFileDescriptor);
3347 return -1;
3348 }
3349 UnixSocket* localSocket = local->get();
3350
3351 N_NOTICE(" -> unix connect: '" << pathname << "'");
3352
3353 SharedPointer<UnixSocketReference> targetReference = acquireSocket(pathname);
3354 if (!targetReference) {
3355 N_NOTICE(" -> unix socket '" << pathname << "' doesn't exist");
3356 return -1;
3357 }
3358 UnixSocket* target = targetReference->get();
3359
3360 if (getType() != SOCK_DGRAM) {
3361 N_NOTICE(" -> connection-oriented");
3362 if (target->getType() != getSocketType()) {
3363 SYSCALL_ERROR(ProtocolWrongType);
3364 return -1;
3365 }
3366 if (target->getState() != UnixSocket::Listening) {
3367 SYSCALL_ERROR(ConnectionRefused);
3368 return -1;
3369 }
3370
3371 // Create the remote for accept() on the server side.
3372 String localPath;
3373 {
3375 localPath = m_LocalPath;
3376 }
3377 UnixSocket* remote =
3378 new UnixSocket(localPath, g_pUnixSocketBacking, nullptr, nullptr, getSocketType());
3379
3380 // Pair first so accept can never observe an endpoint before its peer
3381 // exists. addSocket activates and queues the connection atomically;
3382 // accept only transfers ownership of the queued endpoint.
3383 if (!localSocket->bind(remote, false)) {
3384 delete remote;
3385 SYSCALL_ERROR(IsConnected);
3386 return -1;
3387 }
3388 registerPeer(localSocket, remote, target);
3389 if (!target->addSocket(remote)) {
3390 unregisterPeer(localSocket, remote);
3391 remote->failConnection();
3392 delete remote;
3393 SYSCALL_ERROR(ConnectionRefused);
3394 return -1;
3395 }
3396 notifySocket(target, ReadyRead);
3397 notifyReadiness(ReadyWrite);
3398 N_NOTICE(" -> stream connected and queued");
3399 } else {
3400 if (target->getType() != UnixSocket::Datagram) {
3401 SYSCALL_ERROR(ProtocolWrongType);
3402 return -1;
3403 }
3404 if (target->getState() == UnixSocket::Closed) {
3405 SYSCALL_ERROR(ConnectionRefused);
3406 return -1;
3407 }
3408 N_NOTICE(" -> dgram");
3409 }
3410
3412 {
3414 previousRemote = pedigree_std::move(m_RemoteEndpoint);
3415 m_RemoteEndpoint = pedigree_std::move(targetReference);
3416 m_RemotePath = pathname;
3417 }
3418
3419 if (getType() == SOCK_DGRAM) {
3420 notifyReadiness(ReadyWrite);
3421 }
3422
3423 N_NOTICE(" -> remote is now " << pathname);
3424
3425 if (getType() != SOCK_DGRAM && !isBlocking()) {
3426 SYSCALL_ERROR(InProgress);
3427 return -1;
3428 }
3429
3430 return 0;
3431}
3432
3433ssize_t UnixSocketSyscalls::sendto_msg(const struct msghdr* msghdr,
3434 const SharedPointer<SocketRights>& rights) {
3435 N_NOTICE("UnixSocketSyscalls::sendto_msg");
3436
3438 SharedPointer<UnixSocketReference> remoteReference;
3439 String localPath;
3440 {
3442 local = m_LocalEndpoint;
3443 remoteReference = m_RemoteEndpoint;
3444 localPath = m_LocalPath;
3445 }
3446 if (!local) {
3447 SYSCALL_ERROR(BadFileDescriptor);
3448 return -1;
3449 }
3450
3451 UnixSocket* localSocket = local->get();
3452 const bool blocking = isBlocking() && !(msghdr->msg_flags & MSG_DONTWAIT);
3453 if (getType() != SOCK_DGRAM) {
3454 if (localSocket->wasConnected()) {
3455 remoteReference = local->reference();
3456 } else {
3457 remoteReference.reset();
3458 }
3459 }
3460
3461 UnixSocket* remote = remoteReference ? remoteReference->get() : nullptr;
3462 if (getType() != SOCK_DGRAM && !remote) {
3463 const bool closed = localSocket->getState() == UnixSocket::Closed;
3464 N_NOTICE(" -> " << (closed ? "closed" : "not connected"));
3465 syscallError(closed ? Error::BrokenPipe : Error::NotConnected);
3466 return -1;
3467 }
3468 if (getType() != SOCK_DGRAM && localSocket->writeShutdown()) {
3469 SYSCALL_ERROR(BrokenPipe);
3470 return -1;
3471 }
3472
3473 if (getType() == SOCK_DGRAM && (msghdr->msg_name || !remote)) {
3474 if (!msghdr->msg_name) {
3475 syscallError(EDESTADDRREQ);
3476 N_NOTICE(" -> sendto on unconnected socket with no address");
3477 return -1;
3478 }
3479
3480 String pathname;
3481 if (!unixSocketPath(reinterpret_cast<const struct sockaddr_storage*>(msghdr->msg_name),
3482 msghdr->msg_namelen, pathname, false)) {
3483 return -1;
3484 }
3485
3486 N_NOTICE(" -> unix connect: '" << pathname << "'");
3487
3488 remoteReference = acquireSocket(pathname);
3489 if (!remoteReference) {
3490 N_NOTICE(" -> unix socket '" << pathname << "' doesn't exist");
3491 return -1;
3492 }
3493 remote = remoteReference->get();
3494 }
3495
3496 if (getType() == SOCK_DGRAM && (!remote || remote->getType() != UnixSocket::Datagram ||
3497 remote->getState() == UnixSocket::Closed)) {
3498 syscallError(remote && remote->getType() != UnixSocket::Datagram ? Error::ProtocolWrongType
3499 : Error::ConnectionRefused);
3500 return -1;
3501 }
3502
3503 if (getType() == SOCK_STREAM) {
3504 bool hasPayload = false;
3505 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
3506 if (msghdr->msg_iov[i].iov_len) {
3507 hasPayload = true;
3508 break;
3509 }
3510 }
3511 if (!hasPayload) {
3512 return 0;
3513 }
3514 }
3515
3516 N_NOTICE(" -> transmitting!");
3517
3518 uint64_t numWritten = 0;
3519 bool completedWrite = false;
3520 bool interrupted = false;
3521 int packetError = 0;
3522 if (getType() != SOCK_STREAM) {
3523 size_t datagramLength = 0;
3524 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
3525 if (msghdr->msg_iov[i].iov_len > static_cast<size_t>(SSIZE_MAX) - datagramLength) {
3526 SYSCALL_ERROR(InvalidArgument);
3527 return -1;
3528 }
3529 datagramLength += msghdr->msg_iov[i].iov_len;
3530 }
3531
3532 UniqueArray<uint8_t> datagram;
3533 const void* buffer = nullptr;
3534 if (datagramLength) {
3535 datagram = UniqueArray<uint8_t>::allocate(datagramLength);
3536 if (!datagram) {
3537 SYSCALL_ERROR(OutOfMemory);
3538 return -1;
3539 }
3540 size_t offset = 0;
3541 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
3542 MemoryCopy(datagram.get() + offset, msghdr->msg_iov[i].iov_base,
3543 msghdr->msg_iov[i].iov_len);
3544 offset += msghdr->msg_iov[i].iov_len;
3545 }
3546 buffer = datagram.get();
3547 }
3548
3549 if (getType() == SOCK_SEQPACKET) {
3550 completedWrite = localSocket->sendPacket(datagramLength, reinterpret_cast<uintptr_t>(buffer),
3551 blocking, rights, &packetError);
3552 } else {
3553 completedWrite =
3554 remote->sendDatagram(datagramLength, reinterpret_cast<uintptr_t>(buffer), blocking,
3555 reinterpret_cast<uintptr_t>(localPath.cstr()), rights, &packetError);
3556 }
3557 numWritten = completedWrite ? datagramLength : 0;
3558 } else {
3559 numWritten = localSocket->sendStream(msghdr->msg_iov, static_cast<size_t>(msghdr->msg_iovlen),
3560 isBlocking(), rights, &interrupted);
3561 completedWrite = numWritten;
3562 }
3563 if (completedWrite) {
3564 if (getType() != SOCK_DGRAM) {
3565 notifyPeer(localSocket, ReadyRead);
3566 } else {
3567 notifySocket(remote, ReadyRead);
3568 }
3569 }
3570 if (!completedWrite) {
3571 if (packetError) {
3572 syscallError(packetError);
3573 return -1;
3574 }
3575 if (interrupted) {
3576 SYSCALL_ERROR(Interrupted);
3577 N_NOTICE(" -> -1 (EINTR)");
3578 return -1;
3579 }
3580
3581 if (getType() != SOCK_DGRAM &&
3582 (localSocket->getState() == UnixSocket::Closed || localSocket->writeShutdown())) {
3583 SYSCALL_ERROR(BrokenPipe);
3584 N_NOTICE(" -> -1 (EPIPE)");
3585 return -1;
3586 }
3587
3588 if (!blocking) {
3589 SYSCALL_ERROR(NoMoreProcesses);
3590 N_NOTICE(" -> -1 (EAGAIN)");
3591 return -1;
3592 }
3593 }
3594 N_NOTICE(" -> " << numWritten);
3595 return numWritten;
3596}
3597
3598ssize_t UnixSocketSyscalls::recvfrom_msg(struct msghdr* msghdr,
3600 if (rights) {
3601 rights->reset();
3602 }
3603
3604 const int inputFlags = msghdr->msg_flags;
3605 const bool blocking = isBlocking() && !(inputFlags & MSG_DONTWAIT);
3606#ifdef MSG_TRUNC
3607 if ((inputFlags & MSG_TRUNC) && getType() == SOCK_STREAM) {
3608 SYSCALL_ERROR(OperationNotSupported);
3609 return -1;
3610 }
3611#endif
3612
3613 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
3614 if (!local) {
3615 SYSCALL_ERROR(BadFileDescriptor);
3616 return -1;
3617 }
3618 UnixSocket* localSocket = local->get();
3619 if (getType() == SOCK_SEQPACKET && !localSocket->wasConnected()) {
3620 SYSCALL_ERROR(NotConnected);
3621 return -1;
3622 }
3623
3624 String remote;
3625 uint64_t numRead = 0;
3626 uint64_t datagramLength = 0;
3627 bool consumedDatagram = false;
3628 bool interrupted = false;
3629 if (getType() != SOCK_STREAM) {
3630 size_t datagramCapacity = 0;
3631 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen); ++i) {
3632 if (msghdr->msg_iov[i].iov_len > static_cast<size_t>(SSIZE_MAX) - datagramCapacity) {
3633 SYSCALL_ERROR(InvalidArgument);
3634 return -1;
3635 }
3636 datagramCapacity += msghdr->msg_iov[i].iov_len;
3637 }
3638
3639 UniqueArray<uint8_t> datagram;
3640 void* buffer = nullptr;
3641 if (datagramCapacity) {
3642 datagram = UniqueArray<uint8_t>::allocate(datagramCapacity);
3643 if (!datagram) {
3644 SYSCALL_ERROR(OutOfMemory);
3645 return -1;
3646 }
3647 buffer = datagram.get();
3648 }
3649
3650 SharedPointer<SocketRights> receivedRights;
3651 if (getType() == SOCK_SEQPACKET) {
3652 consumedDatagram = localSocket->receivePacket(
3653 datagramCapacity, reinterpret_cast<uintptr_t>(buffer), blocking, receivedRights, numRead,
3654 datagramLength, &interrupted);
3655 } else {
3656 consumedDatagram =
3657 localSocket->receiveDatagram(datagramCapacity, reinterpret_cast<uintptr_t>(buffer),
3658 blocking, remote, receivedRights, numRead, datagramLength);
3659 }
3660 if (rights) {
3661 *rights = receivedRights;
3662 }
3663 if (consumedDatagram && numRead) {
3664 size_t offset = 0;
3665 for (size_t i = 0; i < static_cast<size_t>(msghdr->msg_iovlen) && offset < numRead; ++i) {
3666 const size_t remaining = static_cast<size_t>(numRead) - offset;
3667 const size_t amount =
3668 msghdr->msg_iov[i].iov_len < remaining ? msghdr->msg_iov[i].iov_len : remaining;
3669 MemoryCopy(msghdr->msg_iov[i].iov_base, datagram.get() + offset, amount);
3670 offset += amount;
3671 }
3672 }
3673 } else {
3674 numRead = localSocket->receiveStream(msghdr->msg_iov, static_cast<size_t>(msghdr->msg_iovlen),
3675 blocking, rights, &interrupted);
3676 }
3677
3678#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
3679 UnixEndpointReceiveLeaseHook leaseHook =
3680 __atomic_load_n(&g_UnixEndpointReceiveLeaseHook, __ATOMIC_ACQUIRE);
3681 if (leaseHook) {
3682 leaseHook();
3683 }
3684#endif
3685
3686 // The receive attempt has established the socket's new readable level,
3687 // including a successful zero-length datagram and a drain to empty.
3688 notifyReadiness(ReadyRead);
3689
3690 if ((numRead || consumedDatagram) && getType() != SOCK_DGRAM) {
3691 // Consuming an incoming record or stream bytes frees the peer's outgoing
3692 // capacity. The peer rechecks the precise level before reporting POLLOUT.
3693 notifyPeer(localSocket, ReadyWrite);
3694 }
3695
3696 if ((numRead || consumedDatagram) && msghdr->msg_name) {
3697 writeUnixSocketAddress(remote, reinterpret_cast<struct sockaddr_storage*>(msghdr->msg_name),
3698 &msghdr->msg_namelen);
3699 }
3700
3701 msghdr->msg_flags = 0;
3702#ifdef MSG_TRUNC
3703 if (consumedDatagram && numRead < datagramLength) {
3704 msghdr->msg_flags |= MSG_TRUNC;
3705 }
3706#endif
3707 if (!numRead && !consumedDatagram) {
3708 if (interrupted) {
3709 SYSCALL_ERROR(Interrupted);
3710 N_NOTICE(" -> -1 (EINTR)");
3711 return -1;
3712 }
3713
3714 if (getType() != SOCK_DGRAM && (localSocket->getState() == UnixSocket::Closed ||
3715 (getType() == SOCK_SEQPACKET && localSocket->readShutdown()))) {
3716 N_NOTICE(" -> 0 (EOF)");
3717 return 0;
3718 }
3719
3720 if (!blocking) {
3721 SYSCALL_ERROR(NoMoreProcesses);
3722 N_NOTICE(" -> -1 (EAGAIN)");
3723 return -1;
3724 }
3725 }
3726
3727#ifdef MSG_TRUNC
3728 if (consumedDatagram && (inputFlags & MSG_TRUNC)) {
3729 N_NOTICE(" -> " << datagramLength);
3730 return datagramLength;
3731 }
3732#endif
3733 N_NOTICE(" -> " << numRead);
3734 return numRead;
3735}
3736
3738 (void)backlog;
3739
3740 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
3741 if (!local) {
3742 SYSCALL_ERROR(BadFileDescriptor);
3743 return -1;
3744 }
3745 UnixSocket* localSocket = local->get();
3746
3747 if (localSocket->getType() == UnixSocket::Datagram) {
3748 SYSCALL_ERROR(OperationNotSupported);
3749 return -1;
3750 }
3751
3753
3754 if (!localSocket->markListening()) {
3755 SYSCALL_ERROR(InvalidArgument);
3756 return -1;
3757 }
3758
3759 return 0;
3760}
3761
3762int UnixSocketSyscalls::bind(const struct sockaddr_storage* address, socklen_t addrlen) {
3763 ResolvedPath parentLease;
3764 String adjusted_pathname;
3765 if (!unixSocketPath(address, addrlen, adjusted_pathname, true)) {
3766 return -1;
3767 }
3768 if (!adjusted_pathname.length()) {
3770 return 0;
3771 }
3772
3773 EndpointMutationGuard mutationGuard(*this);
3774 if (hasLastDescriptorClosed()) {
3775 SYSCALL_ERROR(BadFileDescriptor);
3776 return -1;
3777 }
3778 {
3780 if (!m_LocalEndpoint) {
3781 SYSCALL_ERROR(BadFileDescriptor);
3782 return -1;
3783 }
3784 if (m_LocalPath.length()) {
3785 SYSCALL_ERROR(InvalidArgument);
3786 return -1;
3787 }
3788 }
3789
3790 N_NOTICE(" -> unix bind: '" << adjusted_pathname << "'");
3791
3792 if (isAbstractUnixSocket(adjusted_pathname)) {
3793 UnixSocket* socket =
3794 new UnixSocket(String(), g_pUnixSocketBacking, nullptr, nullptr, getSocketType());
3795 if (!socket) {
3796 SYSCALL_ERROR(OutOfMemory);
3797 return -1;
3798 }
3799
3801 new UnixSocketReference(socket, UnixSocketReferenceOwnership::Heap));
3802 if (!publishAbstractSocket(adjusted_pathname, reference)) {
3803 return -1;
3804 }
3805
3806 replaceLocalEndpoint(reference, &adjusted_pathname, true);
3807 notifyReadiness(ReadyWrite);
3808 return 0;
3809 }
3810
3811 if (adjusted_pathname.endswith('/')) {
3812 // uh, that's a directory
3813 SYSCALL_ERROR(IsADirectory);
3814 return -1;
3815 }
3816
3817 Process* process = Processor::information().getCurrentThread()->getParent();
3818 auto context = process->acquireFilesystemContext();
3819 auto* view = VFS::instance().mountView();
3820 if (!context || !view) {
3821 SYSCALL_ERROR(DoesNotExist);
3822 return -1;
3823 }
3824 FilesystemPathRef parent;
3825 String basename;
3826 if (!view->resolveParent(context, FilesystemPathRef(), adjusted_pathname, parent, basename))
3827 return -1;
3828 parentLease.retain(parent);
3829 if (!parent || parent->provider() != view || !parent->node()->isDirectory()) {
3830 SYSCALL_ERROR(NotADirectory);
3831 return -1;
3832 }
3833 if (!basename.length() || basename == "." || basename == "..") {
3834 SYSCALL_ERROR(AddressInUse);
3835 return -1;
3836 }
3837 if (basename.length() > NAME_MAX) {
3838 SYSCALL_ERROR(NameTooLong);
3839 return -1;
3840 }
3841 File* parentDirectory = parent->node();
3842 if (parentDirectory->getFilesystem()->isReadOnly()) {
3843 SYSCALL_ERROR(ReadOnlyFilesystem);
3844 return -1;
3845 }
3846 if (!VFS::checkAccess(parentDirectory, false, true, true))
3847 return -1;
3848 UnixSocket* socket = new UnixSocket(basename, parentDirectory->getFilesystem(), parentDirectory,
3849 nullptr, getSocketType());
3850 if (!socket) {
3851 SYSCALL_ERROR(OutOfMemory);
3852 return -1;
3853 }
3854 // Establish the descriptor's ownership before publishing the pathname.
3855 // createEphemeral adds the directory's separate ownership on success.
3856 VFS::instance().trackFile(socket);
3857 syscallError(0);
3858 const Directory::AddStatus addStatus =
3859 view->createEphemeral(parent, socket, LandlockAccess::MakeSock);
3860 if (addStatus != Directory::AddStatus::Added) {
3861 const size_t error = Processor::information().getCurrentThread()->getErrno();
3862 socket->releaseVfsReference();
3863 if (error) {
3864 syscallError(error);
3865 } else if (addStatus == Directory::AddStatus::IoError) {
3866 SYSCALL_ERROR(IoError);
3867 } else if (addStatus == Directory::AddStatus::Detached) {
3868 SYSCALL_ERROR(DoesNotExist);
3869 } else {
3870 SYSCALL_ERROR(AddressInUse);
3871 }
3872 return -1;
3873 }
3874 N_NOTICE(" -> basename=" << basename);
3875
3876 // Readers and acceptors retain the old generation outside the state lock.
3877 // Retiring it wakes those operations; destruction waits for their local
3878 // SharedPointer copies to leave scope.
3879 replaceLocalEndpoint(socket, true, &adjusted_pathname);
3880 notifyReadiness(ReadyWrite);
3881
3882 return 0;
3883}
3884
3885int UnixSocketSyscalls::accept(struct sockaddr_storage* address, socklen_t* addrlen, int flags,
3886 DescriptorLease* accepted) {
3887 N_NOTICE("unix accept");
3889 String localPath;
3890 {
3892 local = m_LocalEndpoint;
3893 localPath = m_LocalPath;
3894 }
3895 if (!local) {
3896 SYSCALL_ERROR(BadFileDescriptor);
3897 return -1;
3898 }
3899
3900 UnixSocketSyscalls* obj = createAcceptedSocket();
3901 if (!obj) {
3902 SYSCALL_ERROR(OutOfMemory);
3903 return -1;
3904 }
3905 UnixSocket* remote = local->get()->getSocket(isBlocking());
3906 if (!remote) {
3907 delete obj;
3908 N_NOTICE("accept() failed");
3909 SYSCALL_ERROR(NoMoreProcesses);
3910 return -1;
3911 }
3912
3913 if (remote) {
3914 N_NOTICE("accept() got a socket");
3915
3916 writeUnixSocketAddress(remote->getName(), address, addrlen);
3917
3918 obj->m_NetworkNamespace = m_NetworkNamespace;
3919 obj->m_RemotePath = remote->getName();
3920 obj->replaceLocalEndpoint(remote, false, &localPath);
3921 obj->create();
3922
3923 FileDescriptor* desc = new FileDescriptor;
3925 setSocketDescriptorFlags(desc, flags);
3926
3927 DescriptorLease installed;
3928
3929 const size_t fd = installDescriptor(desc, installed);
3930
3931 if (accepted) {
3932 *accepted = pedigree_std::move(installed);
3933 }
3934 obj->associate(desc);
3935
3936 return static_cast<int>(fd);
3937 }
3938
3939 return -1;
3940}
3941
3942UnixSocketSyscalls* UnixSocketSyscalls::createAcceptedSocket() {
3943 return new UnixSocketSyscalls(m_Domain, m_Type, m_Protocol);
3944}
3945
3946int UnixSocketSyscalls::shutdown(int how) {
3947 N_NOTICE("UnixSocketSyscalls::shutdown");
3948 if (how != SHUT_RD && how != SHUT_WR && how != SHUT_RDWR) {
3949 SYSCALL_ERROR(InvalidArgument);
3950 return -1;
3951 }
3952
3953 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
3954 if (!local) {
3955 SYSCALL_ERROR(BadFileDescriptor);
3956 return -1;
3957 }
3958 UnixSocket* socket = local->get();
3959 if (!socket->shutdown(how)) {
3960 return -1;
3961 }
3962
3963 notifySocket(socket, ReadyRead | ReadyWrite);
3964 notifyPeer(socket, ReadyRead | ReadyWrite);
3965 return 0;
3966}
3967
3968int UnixSocketSyscalls::getpeername(struct sockaddr_storage* address, socklen_t* address_len) {
3969 N_NOTICE("UNIX getpeername");
3971 String remotePath;
3972 {
3974 local = m_LocalEndpoint;
3975 remotePath = m_RemotePath;
3976 }
3977 if (!local) {
3978 SYSCALL_ERROR(BadFileDescriptor);
3979 return -1;
3980 }
3981 if (!local->get()->wasConnected()) {
3982 SYSCALL_ERROR(NotConnected);
3983 return -1;
3984 }
3985
3986 writeUnixSocketAddress(remotePath, address, address_len);
3987
3988 N_NOTICE(" -> " << remotePath);
3989 return 0;
3990}
3991
3992int UnixSocketSyscalls::getsockname(struct sockaddr_storage* address, socklen_t* address_len) {
3993 N_NOTICE("UNIX getsockname");
3994 String localPath;
3995 {
3997 if (!m_LocalEndpoint) {
3998 SYSCALL_ERROR(BadFileDescriptor);
3999 return -1;
4000 }
4001 localPath = m_LocalPath;
4002 }
4003 writeUnixSocketAddress(localPath, address, address_len);
4004
4005 N_NOTICE(" -> " << localPath);
4006 return 0;
4007}
4008
4009int UnixSocketSyscalls::setsockopt(int level, int optname, const void* optvalue, socklen_t optlen) {
4010 if (level == SOL_SOCKET && optname == SO_REUSEADDR) {
4011 if (optlen < sizeof(int)) {
4012 SYSCALL_ERROR(InvalidArgument);
4013 return -1;
4014 }
4015
4016 // Local pathname sockets have no TIME_WAIT address state, but Go's
4017 // listener setup applies SO_REUSEADDR to every Linux socket.
4018 (void)*reinterpret_cast<const int*>(optvalue);
4019 return 0;
4020 }
4021
4022 SYSCALL_ERROR(ProtocolNotAvailable);
4023 return -1;
4024}
4025
4026int UnixSocketSyscalls::getsockopt(int level, int optname, void* optvalue, socklen_t* optlen) {
4027 if (level == SOL_SOCKET) {
4028 if (optname == SO_TYPE || optname == SO_ERROR) {
4029 if (*optlen < sizeof(int)) {
4030 SYSCALL_ERROR(InvalidArgument);
4031 return -1;
4032 }
4033
4034 int value = getType();
4035 if (optname == SO_ERROR) {
4036 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4037 if (!local) {
4038 SYSCALL_ERROR(BadFileDescriptor);
4039 return -1;
4040 }
4041 UnixSocket* localSocket = local->get();
4042 const UnixSocket::SocketState state = localSocket->getState();
4043 if (localSocket->wasConnected()) {
4044 value = 0;
4045 } else if (state == UnixSocket::Connecting) {
4046 value = Error::InProgress;
4047 } else if (state == UnixSocket::Closed) {
4048 value = Error::ConnectionRefused;
4049 } else {
4050 value = Error::NotConnected;
4051 }
4052 }
4053
4054 *reinterpret_cast<int*>(optvalue) = value;
4055 *optlen = sizeof(value);
4056 return 0;
4057 } else if (optname == SO_PEERCRED) {
4058 N_NOTICE(" -> SO_PEERCRED");
4059 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4060 if (!local) {
4061 SYSCALL_ERROR(BadFileDescriptor);
4062 return -1;
4063 }
4064 UnixSocket* localSocket = local->get();
4065 if (!localSocket->wasConnected()) {
4066 SYSCALL_ERROR(NotConnected);
4067 return -1;
4068 }
4069 if (*optlen < sizeof(struct ucred)) {
4070 SYSCALL_ERROR(InvalidArgument);
4071 return -1;
4072 }
4073
4074 // get credentials of other side of this socket
4075 struct ucred* targetCreds = reinterpret_cast<struct ucred*>(optvalue);
4076 struct ucred sourceCreds = localSocket->getPeerCredentials();
4077
4078 N_NOTICE(" --> pid=" << Dec << sourceCreds.pid);
4079 N_NOTICE(" --> uid=" << Dec << sourceCreds.uid);
4080 N_NOTICE(" --> gid=" << Dec << sourceCreds.gid);
4081
4082 *targetCreds = sourceCreds;
4083 *optlen = sizeof(sourceCreds);
4084
4085 return 0;
4086 }
4087 }
4088
4089 SYSCALL_ERROR(ProtocolNotAvailable);
4090 return -1;
4091}
4092
4093bool UnixSocketSyscalls::canPoll() const {
4094 return static_cast<bool>(acquireLocalEndpoint());
4095}
4096
4097ReadyMask UnixSocketSyscalls::queryReady(bool reading, bool writing) {
4098 // Endpoint generations are released at final descriptor close, while the
4099 // NetworkSyscalls wrapper can remain retained by an epoll watch.
4100 EndpointReadinessGuard query(*this);
4101 if (!query) {
4102 return ReadyInvalid | ReadyHangup;
4103 }
4104
4105 if (hasLastDescriptorClosed()) {
4106 return ReadyInvalid | ReadyHangup;
4107 }
4108
4109 SharedPointer<UnixSocketGeneration> endpoint = acquireLocalEndpoint();
4110 if (!endpoint) {
4111 return ReadyInvalid;
4112 }
4113 UnixSocket* local = endpoint->get();
4114
4115 const UnixSocket::SocketState state = local->getState();
4116 if (state == UnixSocket::Closed) {
4117 ReadyMask ready = ReadyHangup;
4118 if (reading) {
4119 ready |= ReadyRead | ReadyReadHangup;
4120 }
4121 if (!local->wasConnected()) {
4122 ready |= ReadyError;
4123 }
4124 return ready | pendingReceiveReadiness();
4125 }
4126
4127 ReadyMask ready = ReadyNone;
4128 if (reading && local->select(false, 0)) {
4129 ready |= ReadyRead;
4130 if (local->readShutdown()) {
4131 ready |= ReadyReadHangup;
4132 }
4133 }
4134
4135 if (writing) {
4136 if (getType() == SOCK_DGRAM) {
4137 // Datagram POLLOUT describes the local send path. A later sendto may
4138 // still race a particular destination becoming full.
4139 ready |= ReadyWrite;
4140 } else if (local->select(true, 0)) {
4141 ready |= ReadyWrite;
4142 }
4143 }
4144
4145 return ready | pendingReceiveReadiness();
4146}
4147
4149 EndpointReadinessGuard query(*this);
4150 if (!query || hasLastDescriptorClosed()) {
4151 return ReadinessGenerations();
4152 }
4153
4154 SharedPointer<UnixSocketGeneration> endpoint = acquireLocalEndpoint();
4155 return withReceiveErrorGeneration(endpoint ? endpoint->get()->readinessGenerations()
4157}
4158
4159bool UnixSocketSyscalls::poll(bool& read, bool& write, bool& error, Semaphore* waiter) {
4160 SharedPointer<UnixSocketGeneration> endpoint = acquireLocalEndpoint();
4161 UnixSocket* local = endpoint ? endpoint->get() : nullptr;
4162 const bool checkRead = read;
4163 const bool checkWrite = write;
4164 read = false;
4165 write = false;
4166 error = pendingReceiveReadiness();
4167
4168 if (!local) {
4169 error = true;
4170 return true;
4171 }
4172
4173 const UnixSocket::SocketState state = local->getState();
4174 if (state == UnixSocket::Closed) {
4175 // The poll interface cannot express POLLHUP separately. POLLERR wakes
4176 // writers, while readable lets readers drain buffered data then see
4177 // persistent EOF.
4178 read = checkRead;
4179 error = true;
4180 return true;
4181 }
4182
4183 bool ok = error;
4184 if (checkRead) {
4185 read = local->select(false, 0);
4186 ok = ok || read;
4187 }
4188
4189 if (checkWrite) {
4190 write = local->select(true, 0);
4191 ok = ok || write;
4192 }
4193
4194 if (waiter && !ok) {
4195 local->addWaiter(waiter, checkRead, checkWrite);
4196 }
4197
4198 return ok;
4199}
4200
4201void UnixSocketSyscalls::unPoll(Semaphore* waiter) {
4202 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4203 if (local) {
4204 local->get()->removeWaiter(waiter);
4205 }
4206}
4207
4208bool UnixSocketSyscalls::monitor(Thread* pThread, Event* pEvent) {
4209 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4210 if (!local) {
4211 return false;
4212 }
4213
4214 local->get()->addWaiter(pThread, pEvent);
4215 return true;
4216}
4217
4218bool UnixSocketSyscalls::unmonitor(Event* pEvent) {
4219 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4220 if (!local) {
4221 return false;
4222 }
4223
4224 local->get()->removeWaiter(pEvent);
4225 return true;
4226}
4227
4229 if (!other || other == this) {
4230 return false;
4231 }
4232
4233 UnixSocketSyscalls* firstMutation = this;
4234 UnixSocketSyscalls* secondMutation = other;
4235 if (reinterpret_cast<uintptr_t>(&firstMutation->m_EndpointMutationLock) >
4236 reinterpret_cast<uintptr_t>(&secondMutation->m_EndpointMutationLock)) {
4237 UnixSocketSyscalls* temporary = firstMutation;
4238 firstMutation = secondMutation;
4239 secondMutation = temporary;
4240 }
4241 EndpointMutationPairGuard mutationGuard(*firstMutation, *secondMutation);
4242
4243 if (hasLastDescriptorClosed() || other->hasLastDescriptorClosed()) {
4244 return false;
4245 }
4246
4247 SharedPointer<UnixSocketGeneration> local = acquireLocalEndpoint();
4248 SharedPointer<UnixSocketGeneration> otherLocal = other->acquireLocalEndpoint();
4249 if (!local || !otherLocal) {
4250 return false;
4251 }
4252
4253 UnixSocket* localSocket = local->get();
4254 UnixSocket* otherSocket = otherLocal->get();
4255 if (!localSocket->bind(otherSocket)) {
4256 return false;
4257 }
4258
4259 registerPeer(localSocket, otherSocket);
4260
4261 // make sure both sides can use the socket
4262 otherSocket->acknowledgeBind();
4263
4265 SharedPointer<UnixSocketReference> otherPreviousRemote;
4266 {
4268 previousRemote = pedigree_std::move(m_RemoteEndpoint);
4269 m_RemoteEndpoint = otherLocal->reference();
4270 }
4271 {
4272 ConstexprLockGuard<Mutex, THREADS> guard(other->m_EndpointStateLock);
4273 otherPreviousRemote = pedigree_std::move(other->m_RemoteEndpoint);
4274 other->m_RemoteEndpoint = local->reference();
4275 }
4276 notifyReadiness(ReadyWrite);
4277 other->notifyReadiness(ReadyWrite);
4278 return true;
4279}
4280
4281UnixSocket::SocketType UnixSocketSyscalls::getSocketType() const {
4282 if (getType() == SOCK_STREAM) {
4283 return UnixSocket::Streaming;
4284 }
4285 if (getType() == SOCK_SEQPACKET) {
4286 return UnixSocket::SequencedPacket;
4287 }
4288
4289 return UnixSocket::Datagram;
4290}
4291
4292#if HOSTED && PEDIGREE_HOSTED_SMOKE_TESTS
4293namespace {
4294class UnixEndpointLifetimeProbe : public UnixSocket {
4295 public:
4296 explicit UnixEndpointLifetimeProbe(Atomic<size_t>& destructions)
4297 : UnixSocket(String(), nullptr, nullptr, nullptr, UnixSocket::Datagram),
4298 m_Destructions(destructions) {}
4299
4300 ~UnixEndpointLifetimeProbe() override {
4301 m_Destructions += 1;
4302 }
4303
4304 private:
4305 Atomic<size_t>& m_Destructions;
4306};
4307
4308struct UnixEndpointReplacementContext {
4309 explicit UnixEndpointReplacementContext(UnixSocketSyscalls* socket)
4310 : socket(socket),
4311 receiver(nullptr),
4312 continueReceive(0, false),
4313 entered(0),
4314 leaseHeld(0),
4315 returned(0),
4316 result(-2) {}
4317
4318 UnixSocketSyscalls* socket;
4319 Thread* receiver;
4320 Semaphore continueReceive;
4321 Atomic<size_t> entered;
4322 Atomic<size_t> leaseHeld;
4323 Atomic<size_t> returned;
4324 Atomic<ssize_t> result;
4325};
4326
4327UnixEndpointReplacementContext* g_UnixEndpointReplacementContext = nullptr;
4328
4329void holdRetiredUnixEndpointLease() {
4330 UnixEndpointReplacementContext* context = g_UnixEndpointReplacementContext;
4331 if (!context || Processor::information().getCurrentThread() != context->receiver) {
4332 return;
4333 }
4334
4335 context->leaseHeld += 1;
4336 context->continueReceive.acquire();
4337}
4338
4339int blockedUnixEndpointReceive(void* parameter) {
4340 UnixEndpointReplacementContext* context =
4341 reinterpret_cast<UnixEndpointReplacementContext*>(parameter);
4342 char byte = 0;
4343 struct iovec vector = {&byte, sizeof(byte)};
4344 struct msghdr message = {};
4345 message.msg_iov = &vector;
4346 message.msg_iovlen = 1;
4347
4348 context->entered += 1;
4349 context->result = context->socket->recvfrom_msg(&message, nullptr);
4350 context->returned += 1;
4351 return 0;
4352}
4353} // namespace
4354
4355bool runHostedUnixEndpointLifetimeRegression(Process* process) {
4356 constexpr size_t Attempts = 10000;
4357 Atomic<size_t> destructions(0);
4358 UnixSocketSyscalls socket(AF_UNIX, SOCK_DGRAM, 0);
4359 socket.replaceLocalEndpoint(new UnixEndpointLifetimeProbe(destructions), false);
4360
4361 UnixEndpointReplacementContext context(&socket);
4362 Thread* receiver =
4363 new Thread(process, blockedUnixEndpointReceive, &context, nullptr, false, true, true);
4364 receiver->setName("hosted Unix endpoint generation receive");
4365 context.receiver = receiver;
4366 g_UnixEndpointReplacementContext = &context;
4367 __atomic_store_n(&g_UnixEndpointReceiveLeaseHook, &holdRetiredUnixEndpointLease,
4368 __ATOMIC_RELEASE);
4369 const bool started = receiver->start();
4370
4371 bool blocked = false;
4372 for (size_t attempt = 0; attempt < Attempts && started; ++attempt) {
4373 Thread::WaitDebugInfo info = {};
4374 uintptr_t debugAddress = 0;
4375 if (context.entered == 1 && !context.returned && receiver->getWaitDebugInfo(info) &&
4376 info.queue && info.queued && receiver->getDebugState(debugAddress) == Thread::SemWait) {
4377 blocked = true;
4378 break;
4379 }
4381 }
4382
4383 String replacementPath("hosted-replacement");
4384 socket.replaceLocalEndpoint(new UnixEndpointLifetimeProbe(destructions), false, &replacementPath);
4385 bool leaseHeld = false;
4386 for (size_t attempt = 0; attempt < Attempts && started; ++attempt) {
4387 Thread::WaitDebugInfo info = {};
4388 if (context.leaseHeld == 1 && receiver->getWaitDebugInfo(info) && info.queue && info.queued &&
4389 info.channelOwner == &context.continueReceive) {
4390 leaseHeld = true;
4391 break;
4392 }
4394 }
4395 const bool retainedWhileInUse = leaseHeld && destructions == 0;
4396 context.continueReceive.release();
4397 const bool joined = started && receiver->join();
4398 __atomic_store_n(&g_UnixEndpointReceiveLeaseHook, nullptr, __ATOMIC_RELEASE);
4399 g_UnixEndpointReplacementContext = nullptr;
4400
4401 bool passed = started && blocked && retainedWhileInUse && joined && context.returned == 1 &&
4402 context.result == 0 && destructions == 1 &&
4403 (socket.queryReady(true, true) & ReadyWrite);
4404 socket.lastDescriptorClosed();
4405 passed = passed && destructions == 2;
4406
4407 if (!passed) {
4408 ERROR(
4409 "HOSTED-SYSCALL-TEST: FAIL unix-bind-replacement-lifetime: "
4410 "endpoint replacement freed a blocked receive generation or failed to wake it");
4411 return false;
4412 }
4413
4414 NOTICE("HOSTED-SYSCALL-TEST: PASS unix-bind-replacement-lifetime");
4415 return true;
4416}
4417#endif
Definition Event.h:49
SharedPointer< EpollInstance > epollImpl
Epoll implementation for this descriptor (if it is an epoll object).
void setNetworkImpl(const SharedPointer< NetworkSyscalls > &implementation)
void setFlags(int newFlags)
Set flags, distributing any associated changes as needed.
bool eventFdPublished() const
SharedPointer< EventFd > getEventFdImpl() const
int getStatusFlags() const
Get current status flags.
void setStatusFlags(int newFlags)
Set status flags, distributing any associated changes as needed.
SharedPointer< NetworkSyscalls > networkImpl
Network syscall implementation for this descriptor (if it's a socket).
size_t fd
Descriptor number.
bool networkPublished() const
Definition File.h:75
virtual bool retainVfsReference()
Definition File.cc:910
String getName() const
Definition File.cc:782
virtual void releaseVfsReference()
Definition File.cc:914
virtual bool isSocket() const
Definition File.cc:816
virtual bool isDirectory()
Definition File.cc:804
virtual ReadyMask queryReady(bool reading, bool writing)
Definition File.cc:1054
bool isReadOnly()
Definition Filesystem.h:150
Definition List.h:61
Iterator begin()
Definition List.h:122
Iterator end()
Definition List.h:132
static void netconnCallback(struct netconn *conn, enum netconn_evt evt, uint16_t len)
ReadinessGenerations readinessGenerations() override
virtual int connect(const struct sockaddr_storage *address, socklen_t addrlen)
virtual void lastDescriptorClosed()
virtual int getpeername(struct sockaddr_storage *address, socklen_t *address_len)
virtual int accept(struct sockaddr_storage *address, socklen_t *addrlen, int flags, DescriptorLease *accepted=nullptr)
virtual ReadyMask queryReady(bool reading, bool writing)
virtual bool poll(bool &read, bool &write, bool &error, Semaphore *waiter)
virtual bool create()
Implementation-specific final socket creation logic.
virtual int getsockname(struct sockaddr_storage *address, socklen_t *address_len)
Definition Mutex.h:56
bool isOwnedByCurrentThread() const
Definition Mutex.cc:30
bool beginDescriptorClose()
virtual ReadyMask queryReady(bool reading, bool writing)
MUST_USE_RESULT bool addDescriptorOwner()
virtual void lastDescriptorClosed()
OperationBarrier m_ReadinessNotifications
ReadinessGenerations readinessGenerations() override
void retainDescriptorLifetime(const SharedPointer< NetworkSyscalls > &lifetime)
MUST_USE_RESULT bool tryAcquire(Lease &lease)
bool closeFileDescriptor(size_t fd, const DescriptorLease &descriptor)
static bool copyFromUser(void *destination, const void *source, size_t count, size_t elementSize=1)
size_t installFileDescriptor(FileDescriptor *descriptor, DescriptorLease &lease, size_t minimum=0)
static bool checkUserBuffer(uintptr_t addr, size_t count, size_t elementSize, size_t flags, size_t *extent=nullptr)
static bool copyToUser(void *destination, const void *source, size_t count, size_t elementSize=1)
static bool checkAddress(uintptr_t addr, size_t extent, size_t flags)
Process * getParent()
Definition Process.h:620
static ProcessorInformation & information()
void notifyReadiness(ReadyMask mask)
Definition Readiness.cc:201
void closeReadiness(ReadyMask mask=ReadyInvalid|ReadyHangup)
Definition Readiness.cc:208
static Scheduler & instance()
Definition Scheduler.h:96
void yield()
Definition Scheduler.cc:236
void release(size_t n=1)
Definition Semaphore.cc:549
bool tryAcquire(size_t n=1)
Definition Semaphore.cc:484
bool acquire(size_t n=1, size_t timeoutSecs=0, size_t timeoutUsecs=0)
Definition Semaphore.cc:355
T * get() const
static bool create(size_t descriptorCount, SharedPointer< SocketRights > &rights)
bool endswith(const char c) const
Definition String.cc:654
bool getWaitDebugInfo(WaitDebugInfo &info)
Definition Thread.cc:3163
bool join()
Definition Thread.cc:2746
DebugState getDebugState(uintptr_t &address)
Definition Thread.h:574
bool start()
Definition Thread.cc:751
An iterator applicable for many data structures.
Definition Iterator.h:147
A key/value dictionary.
Definition Tree.h:33
Iterator begin()
Definition Tree.h:402
void remove(const K &key)
Definition Tree.h:301
E lookup(const K &key) const
Definition Tree.h:193
void insert(const K &key, const E &value)
Definition Tree.h:149
Iterator end()
Definition Tree.h:427
virtual int listen(int backlog)
virtual void lastDescriptorClosed()
ReadinessGenerations readinessGenerations() override
bool pairWith(UnixSocketSyscalls *other)
virtual int bind(const struct sockaddr_storage *address, socklen_t addrlen)
virtual bool create()
Implementation-specific final socket creation logic.
virtual ReadyMask queryReady(bool reading, bool writing)
uint64_t sendStream(uint64_t size, uintptr_t buffer, bool bCanBlock, const SharedPointer< SocketRights > &rights, bool *interrupted=nullptr)
bool sendDatagram(uint64_t size, uintptr_t buffer, bool bCanBlock, uintptr_t source, const SharedPointer< SocketRights > &rights, int *error=nullptr)
uint64_t receiveStream(uint64_t size, uintptr_t buffer, bool bCanBlock, SharedPointer< SocketRights > *rights, bool *interrupted=nullptr)
virtual int select(bool bWriting=false, int timeout=0)
bool receiveDatagram(uint64_t size, uintptr_t buffer, bool bCanBlock, String &from, SharedPointer< SocketRights > &rights, uint64_t &bytesRead, uint64_t &datagramLength)
bool sendPacket(uint64_t size, uintptr_t buffer, bool bCanBlock, const SharedPointer< SocketRights > &rights, int *error=nullptr)
static bool checkAccess(File *pFile, bool bRead, bool bWrite, bool bExecute)
Definition VFS.cc:1502
static VFS & instance()
Definition VFS.cc:311
void trackFile(File *pFile)
Track a File object that exists. It is necessary to keep track of File objects, or at least those tha...
Definition VFS.cc:1625
s8_t err_t
Definition err.h:76
@ ERR_IF
Definition err.h:106
@ ERR_OK
Definition err.h:82
@ ERR_CLSD
Definition err.h:113
@ ERR_VAL
Definition err.h:94
@ IPADDR_TYPE_V4
Definition ip_addr.h:75
@ Dec
Definition Log.h:126
@ Hex
Definition Log.h:124
Iterator erase(Iterator &Iter)
Definition List.h:352
void pushBack(const T &value)
Definition List.h:216
u8_t pbuf_free(struct pbuf *p)
Definition pbuf.c:734
u16_t pbuf_copy_partial(const struct pbuf *buf, void *dataptr, u16_t len, u16_t offset)
Definition pbuf.c:1034
#define INADDR_LOOPBACK
Definition inet.h:92
#define ip_set_option(pcb, opt)
Definition ip.h:236
#define ip_get_option(pcb, opt)
Definition ip.h:234
#define ip_reset_option(pcb, opt)
Definition ip.h:238
Definition ip.h:108
Definition pbuf.h:161
u16_t tot_len
Definition pbuf.h:175