Please note that diffs are not public domain; they are subject to the copyright notices on the relevant files. =================================================================== RCS file: /ftp/cvs/cvsroot/src/sys/kern/uipc_socket.c,v rcsdiff: /ftp/cvs/cvsroot/src/sys/kern/uipc_socket.c,v: warning: Unknown phrases like `commitid ...;' are present. retrieving revision 1.89 retrieving revision 1.97.2.2 diff -u -p -r1.89 -r1.97.2.2 --- src/sys/kern/uipc_socket.c 2003/09/15 00:22:20 1.89 +++ src/sys/kern/uipc_socket.c 2005/10/31 13:37:33 1.97.2.2 @@ -1,4 +1,4 @@ -/* $NetBSD: uipc_socket.c,v 1.89 2003/09/15 00:22:20 christos Exp $ */ +/* $NetBSD: uipc_socket.c,v 1.97.2.2 2005/10/31 13:37:33 tron Exp $ */ /*- * Copyright (c) 2002 The NetBSD Foundation, Inc. @@ -68,7 +68,7 @@ */ #include -__KERNEL_RCSID(0, "$NetBSD: uipc_socket.c,v 1.89 2003/09/15 00:22:20 christos Exp $"); +__KERNEL_RCSID(0, "$NetBSD: uipc_socket.c,v 1.97.2.2 2005/10/31 13:37:33 tron Exp $"); #include "opt_sock_counters.h" #include "opt_sosend_loan.h" @@ -126,6 +126,10 @@ void soinit(void) { + /* Set the initial adjusted socket buffer size. */ + if (sb_max_set(sb_max)) + panic("bad initial sb_max value: %lu\n", sb_max); + pool_init(&socket_pool, sizeof(struct socket), 0, 0, 0, "sockpl", NULL); @@ -143,6 +147,7 @@ int use_sosend_loan = 0; int use_sosend_loan = 1; #endif +struct simplelock so_pendfree_slock = SIMPLELOCK_INITIALIZER; struct mbuf *so_pendfree; #ifndef SOMAXKVA @@ -156,40 +161,104 @@ int sokvawaiters; #define SOCK_LOAN_CHUNK 65536 static size_t sodopendfree(struct socket *); +static size_t sodopendfreel(struct socket *); +static __inline void sokvareserve(struct socket *, vsize_t); +static __inline void sokvaunreserve(vsize_t); -vaddr_t -sokvaalloc(vsize_t len, struct socket *so) +static __inline void +sokvareserve(struct socket *so, vsize_t len) { - vaddr_t lva; int s; + s = splvm(); + simple_lock(&so_pendfree_slock); while (socurkva + len > somaxkva) { - if (sodopendfree(so)) + size_t freed; + + /* + * try to do pendfree. + */ + + freed = sodopendfreel(so); + + /* + * if some kva was freed, try again. + */ + + if (freed) continue; + SOSEND_COUNTER_INCR(&sosend_kvalimit); - s = splvm(); sokvawaiters++; - (void) tsleep(&socurkva, PVM, "sokva", 0); + (void) ltsleep(&socurkva, PVM, "sokva", 0, &so_pendfree_slock); sokvawaiters--; - splx(s); } + socurkva += len; + simple_unlock(&so_pendfree_slock); + splx(s); +} + +static __inline void +sokvaunreserve(vsize_t len) +{ + int s; + + s = splvm(); + simple_lock(&so_pendfree_slock); + socurkva -= len; + if (sokvawaiters) + wakeup(&socurkva); + simple_unlock(&so_pendfree_slock); + splx(s); +} + +/* + * sokvaalloc: allocate kva for loan. + */ + +vaddr_t +sokvaalloc(vsize_t len, struct socket *so) +{ + vaddr_t lva; + + /* + * reserve kva. + */ + + sokvareserve(so, len); + + /* + * allocate kva. + */ lva = uvm_km_valloc_wait(kernel_map, len); - if (lva == 0) + if (lva == 0) { + sokvaunreserve(len); return (0); - socurkva += len; + } return lva; } +/* + * sokvafree: free kva for loan. + */ + void sokvafree(vaddr_t sva, vsize_t len) { + /* + * free kva. + */ + uvm_km_free(kernel_map, sva, len); - socurkva -= len; - if (sokvawaiters) - wakeup(&socurkva); + + /* + * unreserve kva. + */ + + sokvaunreserve(len); } static void @@ -224,63 +293,91 @@ sodoloanfree(struct vm_page **pgs, caddr static size_t sodopendfree(struct socket *so) { - struct mbuf *m; - size_t rv = 0; int s; + size_t rv; s = splvm(); + simple_lock(&so_pendfree_slock); + rv = sodopendfreel(so); + simple_unlock(&so_pendfree_slock); + splx(s); - for (;;) { - m = so_pendfree; - if (m == NULL) - break; - so_pendfree = m->m_next; - splx(s); + return rv; +} - rv += m->m_ext.ext_size; - sodoloanfree((m->m_flags & M_EXT_PAGES) ? - m->m_ext.ext_pgs : NULL, m->m_ext.ext_buf, - m->m_ext.ext_size); - s = splvm(); - pool_cache_put(&mbpool_cache, m); - } +/* + * sodopendfreel: free mbufs on "pendfree" list. + * unlock and relock so_pendfree_slock when freeing mbufs. + * + * => called with so_pendfree_slock held. + * => called at splvm. + */ + +static size_t +sodopendfreel(struct socket *so) +{ + size_t rv = 0; + + LOCK_ASSERT(simple_lock_held(&so_pendfree_slock)); for (;;) { - m = so->so_pendfree; + struct mbuf *m; + struct mbuf *next; + + m = so_pendfree; if (m == NULL) break; - so->so_pendfree = m->m_next; - splx(s); + so_pendfree = NULL; + simple_unlock(&so_pendfree_slock); + /* XXX splx */ + + for (; m != NULL; m = next) { + next = m->m_next; + + rv += m->m_ext.ext_size; + sodoloanfree((m->m_flags & M_EXT_PAGES) ? + m->m_ext.ext_pgs : NULL, m->m_ext.ext_buf, + m->m_ext.ext_size); + pool_cache_put(&mbpool_cache, m); + } - rv += m->m_ext.ext_size; - sodoloanfree((m->m_flags & M_EXT_PAGES) ? - m->m_ext.ext_pgs : NULL, m->m_ext.ext_buf, - m->m_ext.ext_size); - s = splvm(); - pool_cache_put(&mbpool_cache, m); + /* XXX splvm */ + simple_lock(&so_pendfree_slock); } - splx(s); return (rv); } void soloanfree(struct mbuf *m, caddr_t buf, size_t size, void *arg) { - struct socket *so = arg; int s; if (m == NULL) { + + /* + * called from MEXTREMOVE. + */ + sodoloanfree(NULL, buf, size); return; } + /* + * postpone freeing mbuf. + * + * we can't do it in interrupt context + * because we need to put kva back to kernel_map. + */ + s = splvm(); - m->m_next = so->so_pendfree; - so->so_pendfree = m; - splx(s); + simple_lock(&so_pendfree_slock); + m->m_next = so_pendfree; + so_pendfree = m; if (sokvawaiters) wakeup(&socurkva); + simple_unlock(&so_pendfree_slock); + splx(s); } static long @@ -431,7 +528,6 @@ solisten(struct socket *so, int backlog) void sofree(struct socket *so) { - struct mbuf *m; if (so->so_pcb || (so->so_state & SS_NOFDREF) == 0) return; @@ -446,11 +542,6 @@ sofree(struct socket *so) } sbrelease(&so->so_snd); sorflush(so); - while ((m = so->so_pendfree) != NULL) { - so->so_pendfree = m->m_next; - m->m_next = so_pendfree; - so_pendfree = m; - } pool_put(&socket_pool, so); } @@ -692,7 +783,7 @@ sosend(struct socket *so, struct mbuf *a if ((atomic && resid > so->so_snd.sb_hiwat) || clen > so->so_snd.sb_hiwat) snderr(EMSGSIZE); - if (space < resid + clen && uio && + if (space < resid + clen && (atomic || space < so->so_snd.sb_lowat || space < clen)) { if (so->so_state & SS_NBIO) snderr(EWOULDBLOCK); @@ -1300,6 +1391,11 @@ sosetopt(struct socket *so, int level, i error = EINVAL; goto bad; } + if (mtod(m, struct linger *)->l_linger < 0 || + mtod(m, struct linger *)->l_linger > (INT_MAX / hz)) { + error = EDOM; + goto bad; + } so->so_linger = mtod(m, struct linger *)->l_linger; /* fall thru... */ @@ -1507,17 +1603,7 @@ sogetopt(struct socket *so, int level, i void sohasoutofband(struct socket *so) { - struct proc *p; - ksiginfo_t ksi; - memset(&ksi, 0, sizeof(ksi)); - ksi.ksi_signo = SIGURG; - ksi.ksi_band = POLLPRI|POLLRDBAND; - ksi.ksi_code = POLL_PRI; - - if (so->so_pgid < 0) - kgsignal(-so->so_pgid, &ksi, so); - else if (so->so_pgid > 0 && (p = pfind(so->so_pgid)) != 0) - kpsignal(p, &ksi, so); + fownsignal(so->so_pgid, SIGURG, POLL_PRI, POLLPRI|POLLRDBAND, so); selwakeup(&so->so_rcv.sb_sel); } @@ -1636,3 +1722,56 @@ soo_kqfilter(struct file *fp, struct kno return (0); } +#include + +static int sysctl_kern_somaxkva(SYSCTLFN_PROTO); + +/* + * sysctl helper routine for kern.somaxkva. ensures that the given + * value is not too small. + * (XXX should we maybe make sure it's not too large as well?) + */ +static int +sysctl_kern_somaxkva(SYSCTLFN_ARGS) +{ + int error, new_somaxkva; + struct sysctlnode node; + int s; + + new_somaxkva = somaxkva; + node = *rnode; + node.sysctl_data = &new_somaxkva; + error = sysctl_lookup(SYSCTLFN_CALL(&node)); + if (error || newp == NULL) + return (error); + + if (new_somaxkva < (16 * 1024 * 1024)) /* sanity */ + return (EINVAL); + + s = splvm(); + simple_lock(&so_pendfree_slock); + somaxkva = new_somaxkva; + wakeup(&socurkva); + simple_unlock(&so_pendfree_slock); + splx(s); + + return (error); +} + +SYSCTL_SETUP(sysctl_kern_somaxkva_setup, "sysctl kern.somaxkva setup") +{ + + sysctl_createv(clog, 0, NULL, NULL, + CTLFLAG_PERMANENT, + CTLTYPE_NODE, "kern", NULL, + NULL, 0, NULL, 0, + CTL_KERN, CTL_EOL); + + sysctl_createv(clog, 0, NULL, NULL, + CTLFLAG_PERMANENT|CTLFLAG_READWRITE, + CTLTYPE_INT, "somaxkva", + SYSCTL_DESCR("Maximum amount of kernel memory to be " + "used for socket buffers"), + sysctl_kern_somaxkva, 0, NULL, 0, + CTL_KERN, KERN_SOMAXKVA, CTL_EOL); +}