1 /* $NetBSD: dispatch.c,v 1.17 2026/08/29 14:55:15 christos Exp $ */ 2 3 /* 4 * Copyright (C) Internet Systems Consortium, Inc. ("ISC") 5 * 6 * SPDX-License-Identifier: MPL-2.0 7 * 8 * This Source Code Form is subject to the terms of the Mozilla Public 9 * License, v. 2.0. If a copy of the MPL was not distributed with this 10 * file, you can obtain one at https://mozilla.org/MPL/2.0/. 11 * 12 * See the COPYRIGHT file distributed with this work for additional 13 * information regarding copyright ownership. 14 */ 15 16 /*! \file */ 17 18 #include <inttypes.h> 19 #include <stdbool.h> 20 #include <stdlib.h> 21 #include <sys/types.h> 22 #include <unistd.h> 23 24 #include <isc/async.h> 25 #include <isc/hash.h> 26 #include <isc/hashmap.h> 27 #include <isc/loop.h> 28 #include <isc/mem.h> 29 #include <isc/mutex.h> 30 #include <isc/net.h> 31 #include <isc/netmgr.h> 32 #include <isc/portset.h> 33 #include <isc/random.h> 34 #include <isc/stats.h> 35 #include <isc/string.h> 36 #include <isc/tid.h> 37 #include <isc/time.h> 38 #include <isc/tls.h> 39 #include <isc/urcu.h> 40 #include <isc/util.h> 41 42 #include <dns/acl.h> 43 #include <dns/dispatch.h> 44 #include <dns/log.h> 45 #include <dns/message.h> 46 #include <dns/stats.h> 47 #include <dns/transport.h> 48 #include <dns/types.h> 49 50 /* 51 * Maximum number of queries to pipeline on a single shared TCP dispatch. 52 * Once reached, the dispatch is removed from the hash table so new queries 53 * get a fresh connection. Can be overridden via 'named -T tcppipelining=N'. 54 */ 55 size_t dns_dispatch_tcppipelining = 256; 56 57 typedef ISC_LIST(dns_dispentry_t) dns_displist_t; 58 59 struct dns_dispatchmgr { 60 /* Unlocked. */ 61 unsigned int magic; 62 isc_refcount_t references; 63 isc_mem_t *mctx; 64 dns_acl_t *blackhole; 65 isc_stats_t *stats; 66 isc_nm_t *nm; 67 68 unsigned int reuse_timeout; 69 70 uint32_t nloops; 71 72 struct cds_lfht **tcps; 73 74 struct cds_lfht *qids; 75 76 in_port_t *v4ports; /*%< available ports for IPv4 */ 77 unsigned int nv4ports; /*%< # of available ports for IPv4 */ 78 in_port_t *v6ports; /*%< available ports for IPv6 */ 79 unsigned int nv6ports; /*%< # of available ports for IPv6 */ 80 }; 81 82 typedef enum { 83 DNS_DISPATCHSTATE_NONE = 0UL, 84 DNS_DISPATCHSTATE_CONNECTING, 85 DNS_DISPATCHSTATE_CONNECTED, 86 DNS_DISPATCHSTATE_CANCELED, 87 } dns_dispatchstate_t; 88 89 struct dns_dispentry { 90 unsigned int magic; 91 isc_refcount_t references; 92 isc_mem_t *mctx; 93 dns_dispatch_t *disp; 94 isc_loop_t *loop; 95 isc_nmhandle_t *handle; /*%< netmgr handle for UDP connection */ 96 dns_dispatchstate_t state; 97 dns_transport_t *transport; 98 isc_tlsctx_cache_t *tlsctx_cache; 99 unsigned int retries; 100 unsigned int timeout; 101 isc_time_t start; 102 isc_sockaddr_t local; 103 isc_sockaddr_t peer; 104 in_port_t port; 105 dns_messageid_t id; 106 dispatch_cb_t connected; 107 dispatch_cb_t sent; 108 dispatch_cb_t response; 109 void *arg; 110 bool reading; 111 isc_result_t result; 112 ISC_LINK(dns_dispentry_t) alink; 113 ISC_LINK(dns_dispentry_t) plink; 114 ISC_LINK(dns_dispentry_t) rlink; 115 116 struct cds_lfht_node ht_node; 117 struct rcu_head rcu_head; 118 }; 119 120 struct dns_dispatch { 121 /* Unlocked. */ 122 unsigned int magic; /*%< magic */ 123 uint32_t tid; 124 isc_socktype_t socktype; 125 isc_refcount_t references; 126 isc_mem_t *mctx; 127 dns_dispatchmgr_t *mgr; /*%< dispatch manager */ 128 isc_nmhandle_t *handle; /*%< netmgr handle for TCP connection */ 129 isc_sockaddr_t local; /*%< local address */ 130 isc_sockaddr_t peer; /*%< peer address (TCP) */ 131 dns_transport_t *transport; /*%< TCP transport parameters */ 132 133 dns_dispatchopt_t options; 134 dns_dispatchstate_t state; 135 dns_dispatchtype_t disptype; 136 137 dns_messageid_t nextid; /*%< next sequential QID for TCP */ 138 139 bool reading; 140 141 dns_displist_t pending; 142 dns_displist_t active; 143 144 uint_fast32_t requests; /*%< how many requests we have */ 145 146 unsigned int timedout; 147 148 struct cds_lfht_node ht_node; 149 struct rcu_head rcu_head; 150 }; 151 152 #define RESPONSE_MAGIC ISC_MAGIC('D', 'r', 's', 'p') 153 #define VALID_RESPONSE(e) ISC_MAGIC_VALID((e), RESPONSE_MAGIC) 154 155 #define DISPATCH_MAGIC ISC_MAGIC('D', 'i', 's', 'p') 156 #define VALID_DISPATCH(e) ISC_MAGIC_VALID((e), DISPATCH_MAGIC) 157 158 #define DNS_DISPATCHMGR_MAGIC ISC_MAGIC('D', 'M', 'g', 'r') 159 #define VALID_DISPATCHMGR(e) ISC_MAGIC_VALID((e), DNS_DISPATCHMGR_MAGIC) 160 161 #if DNS_DISPATCH_TRACE 162 #define dns_dispentry_ref(ptr) \ 163 dns_dispentry__ref(ptr, __func__, __FILE__, __LINE__) 164 #define dns_dispentry_unref(ptr) \ 165 dns_dispentry__unref(ptr, __func__, __FILE__, __LINE__) 166 #define dns_dispentry_attach(ptr, ptrp) \ 167 dns_dispentry__attach(ptr, ptrp, __func__, __FILE__, __LINE__) 168 #define dns_dispentry_detach(ptrp) \ 169 dns_dispentry__detach(ptrp, __func__, __FILE__, __LINE__) 170 ISC_REFCOUNT_TRACE_DECL(dns_dispentry); 171 #else 172 ISC_REFCOUNT_DECL(dns_dispentry); 173 #endif 174 175 /* 176 * The number of attempts to find unique <addr, port, query_id> combination 177 */ 178 #define QID_MAX_TRIES 64 179 180 /* 181 * Initial and minimum QID table sizes. 182 */ 183 #define QIDS_INIT_SIZE (1 << 4) /* Must be power of 2 */ 184 #define QIDS_MIN_SIZE (1 << 4) /* Must be power of 2 */ 185 186 /* 187 * Statics. 188 */ 189 static void 190 dispatchmgr_destroy(dns_dispatchmgr_t *mgr); 191 192 static void 193 udp_recv(isc_nmhandle_t *handle, isc_result_t eresult, isc_region_t *region, 194 void *arg); 195 static void 196 tcp_recv(isc_nmhandle_t *handle, isc_result_t eresult, isc_region_t *region, 197 void *arg); 198 static void 199 dispentry_cancel(dns_dispentry_t *resp, isc_result_t result); 200 static isc_result_t 201 dispatch_createudp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 202 uint32_t tid, dns_dispatch_t **dispp); 203 static void 204 udp_startrecv(isc_nmhandle_t *handle, dns_dispentry_t *resp); 205 static void 206 udp_dispatch_connect(dns_dispatch_t *disp, dns_dispentry_t *resp); 207 static void 208 tcp_startrecv(dns_dispatch_t *disp, dns_dispentry_t *resp); 209 static void 210 tcp_dispatch_getnext(dns_dispatch_t *disp, dns_dispentry_t *resp, 211 int32_t timeout); 212 static void 213 udp_dispatch_getnext(dns_dispentry_t *resp, int32_t timeout); 214 215 static const char * 216 socktype2str(dns_dispentry_t *resp) { 217 dns_transport_type_t transport_type = DNS_TRANSPORT_UDP; 218 dns_dispatch_t *disp = resp->disp; 219 220 if (disp->socktype == isc_socktype_tcp) { 221 if (resp->transport != NULL) { 222 transport_type = 223 dns_transport_get_type(resp->transport); 224 } else { 225 transport_type = DNS_TRANSPORT_TCP; 226 } 227 } 228 229 switch (transport_type) { 230 case DNS_TRANSPORT_UDP: 231 return "UDP"; 232 case DNS_TRANSPORT_TCP: 233 return "TCP"; 234 case DNS_TRANSPORT_TLS: 235 return "TLS"; 236 case DNS_TRANSPORT_HTTP: 237 return "HTTP"; 238 default: 239 return "<unexpected>"; 240 } 241 } 242 243 static const char * 244 state2str(dns_dispatchstate_t state) { 245 switch (state) { 246 case DNS_DISPATCHSTATE_NONE: 247 return "none"; 248 case DNS_DISPATCHSTATE_CONNECTING: 249 return "connecting"; 250 case DNS_DISPATCHSTATE_CONNECTED: 251 return "connected"; 252 case DNS_DISPATCHSTATE_CANCELED: 253 return "canceled"; 254 default: 255 return "<unexpected>"; 256 } 257 } 258 259 static void 260 mgr_log(dns_dispatchmgr_t *mgr, int level, const char *fmt, ...) 261 ISC_FORMAT_PRINTF(3, 4); 262 263 static void 264 mgr_log(dns_dispatchmgr_t *mgr, int level, const char *fmt, ...) { 265 char msgbuf[2048]; 266 va_list ap; 267 268 if (!isc_log_wouldlog(dns_lctx, level)) { 269 return; 270 } 271 272 va_start(ap, fmt); 273 vsnprintf(msgbuf, sizeof(msgbuf), fmt, ap); 274 va_end(ap); 275 276 isc_log_write(dns_lctx, DNS_LOGCATEGORY_DISPATCH, 277 DNS_LOGMODULE_DISPATCH, level, "dispatchmgr %p: %s", mgr, 278 msgbuf); 279 } 280 281 static void 282 inc_stats(dns_dispatchmgr_t *mgr, isc_statscounter_t counter) { 283 if (mgr->stats != NULL) { 284 isc_stats_increment(mgr->stats, counter); 285 } 286 } 287 288 static void 289 dec_stats(dns_dispatchmgr_t *mgr, isc_statscounter_t counter) { 290 if (mgr->stats != NULL) { 291 isc_stats_decrement(mgr->stats, counter); 292 } 293 } 294 295 static void 296 dispatch_log(dns_dispatch_t *disp, int level, const char *fmt, ...) 297 ISC_FORMAT_PRINTF(3, 4); 298 299 static void 300 dispatch_log(dns_dispatch_t *disp, int level, const char *fmt, ...) { 301 char msgbuf[2048]; 302 va_list ap; 303 int r; 304 305 if (!isc_log_wouldlog(dns_lctx, level)) { 306 return; 307 } 308 309 va_start(ap, fmt); 310 r = vsnprintf(msgbuf, sizeof(msgbuf), fmt, ap); 311 if (r < 0) { 312 msgbuf[0] = '\0'; 313 } else if ((unsigned int)r >= sizeof(msgbuf)) { 314 /* Truncated */ 315 msgbuf[sizeof(msgbuf) - 1] = '\0'; 316 } 317 va_end(ap); 318 319 isc_log_write(dns_lctx, DNS_LOGCATEGORY_DISPATCH, 320 DNS_LOGMODULE_DISPATCH, level, "dispatch %p: %s", disp, 321 msgbuf); 322 } 323 324 static void 325 dispentry_log(dns_dispentry_t *resp, int level, const char *fmt, ...) 326 ISC_FORMAT_PRINTF(3, 4); 327 328 static void 329 dispentry_log(dns_dispentry_t *resp, int level, const char *fmt, ...) { 330 char msgbuf[2048]; 331 va_list ap; 332 int r; 333 334 if (!isc_log_wouldlog(dns_lctx, level)) { 335 return; 336 } 337 338 va_start(ap, fmt); 339 r = vsnprintf(msgbuf, sizeof(msgbuf), fmt, ap); 340 if (r < 0) { 341 msgbuf[0] = '\0'; 342 } else if ((unsigned int)r >= sizeof(msgbuf)) { 343 /* Truncated */ 344 msgbuf[sizeof(msgbuf) - 1] = '\0'; 345 } 346 va_end(ap); 347 348 dispatch_log(resp->disp, level, "%s response %p: %s", 349 socktype2str(resp), resp, msgbuf); 350 } 351 352 /*% 353 * Choose a random port number for a dispatch entry. 354 */ 355 static isc_result_t 356 setup_socket(dns_dispatch_t *disp, dns_dispentry_t *resp, 357 const isc_sockaddr_t *dest, in_port_t *portp) { 358 dns_dispatchmgr_t *mgr = disp->mgr; 359 unsigned int nports; 360 in_port_t *ports = NULL; 361 in_port_t port = *portp; 362 363 if (resp->retries++ > 5) { 364 return ISC_R_FAILURE; 365 } 366 367 if (isc_sockaddr_pf(&disp->local) == AF_INET) { 368 nports = mgr->nv4ports; 369 ports = mgr->v4ports; 370 } else { 371 nports = mgr->nv6ports; 372 ports = mgr->v6ports; 373 } 374 if (nports == 0) { 375 return ISC_R_ADDRNOTAVAIL; 376 } 377 378 resp->local = disp->local; 379 resp->peer = *dest; 380 381 if (port == 0) { 382 port = ports[isc_random_uniform(nports)]; 383 isc_sockaddr_setport(&resp->local, port); 384 *portp = port; 385 } 386 resp->port = port; 387 388 return ISC_R_SUCCESS; 389 } 390 391 static uint32_t 392 qid_hash(const dns_dispentry_t *dispentry) { 393 isc_hash32_t hash; 394 395 isc_hash32_init(&hash); 396 397 isc_sockaddr_hash_ex(&hash, &dispentry->peer, true); 398 isc_hash32_hash(&hash, &dispentry->id, sizeof(dispentry->id), true); 399 isc_hash32_hash(&hash, &dispentry->port, sizeof(dispentry->port), true); 400 401 return isc_hash32_finalize(&hash); 402 } 403 404 static int 405 qid_match(struct cds_lfht_node *node, const void *key0) { 406 const dns_dispentry_t *dispentry = 407 caa_container_of(node, dns_dispentry_t, ht_node); 408 const dns_dispentry_t *key = key0; 409 410 return dispentry->id == key->id && dispentry->port == key->port && 411 isc_sockaddr_equal(&dispentry->peer, &key->peer); 412 } 413 414 static void 415 dispentry_destroy_rcu(struct rcu_head *rcu_head) { 416 dns_dispentry_t *resp = caa_container_of(rcu_head, dns_dispentry_t, 417 rcu_head); 418 isc_mem_putanddetach(&resp->mctx, resp, sizeof(*resp)); 419 } 420 421 static void 422 dispentry_destroy(dns_dispentry_t *resp) { 423 dns_dispatch_t *disp = resp->disp; 424 425 /* 426 * We need to call this from here in case there's an external event that 427 * shuts down our dispatch (like ISC_R_SHUTTINGDOWN). 428 */ 429 dispentry_cancel(resp, ISC_R_CANCELED); 430 431 INSIST(disp->requests > 0); 432 disp->requests--; 433 434 resp->magic = 0; 435 436 INSIST(!ISC_LINK_LINKED(resp, plink)); 437 INSIST(!ISC_LINK_LINKED(resp, alink)); 438 INSIST(!ISC_LINK_LINKED(resp, rlink)); 439 440 dispentry_log(resp, ISC_LOG_DEBUG(90), "destroying"); 441 442 if (resp->handle != NULL) { 443 dispentry_log(resp, ISC_LOG_DEBUG(90), 444 "detaching handle %p from %p", resp->handle, 445 &resp->handle); 446 isc_nmhandle_detach(&resp->handle); 447 } 448 449 if (resp->tlsctx_cache != NULL) { 450 isc_tlsctx_cache_detach(&resp->tlsctx_cache); 451 } 452 453 if (resp->transport != NULL) { 454 dns_transport_detach(&resp->transport); 455 } 456 457 dns_dispatch_detach(&disp); /* DISPATCH001 */ 458 459 call_rcu(&resp->rcu_head, dispentry_destroy_rcu); 460 } 461 462 #if DNS_DISPATCH_TRACE 463 ISC_REFCOUNT_TRACE_IMPL(dns_dispentry, dispentry_destroy); 464 #else 465 ISC_REFCOUNT_IMPL(dns_dispentry, dispentry_destroy); 466 #endif 467 468 /* 469 * How long in milliseconds has it been since this dispentry 470 * started reading? 471 */ 472 static unsigned int 473 dispentry_runtime(dns_dispentry_t *resp, const isc_time_t *now) { 474 if (isc_time_isepoch(&resp->start)) { 475 return 0; 476 } 477 478 return isc_time_microdiff(now, &resp->start) / 1000; 479 } 480 481 /* 482 * General flow: 483 * 484 * If I/O result == CANCELED or error, free the buffer. 485 * 486 * If query, free the buffer, restart. 487 * 488 * If response: 489 * Allocate event, fill in details. 490 * If cannot allocate, free buffer, restart. 491 * find target. If not found, free buffer, restart. 492 * if event queue is not empty, queue. else, send. 493 * restart. 494 */ 495 static void 496 udp_recv(isc_nmhandle_t *handle, isc_result_t eresult, isc_region_t *region, 497 void *arg) { 498 dns_dispentry_t *resp = (dns_dispentry_t *)arg; 499 dns_dispatch_t *disp = NULL; 500 dns_messageid_t id; 501 isc_result_t dres; 502 isc_buffer_t source; 503 unsigned int flags; 504 isc_sockaddr_t peer; 505 isc_netaddr_t netaddr; 506 int match, timeout = 0; 507 bool respond = true; 508 isc_time_t now; 509 510 REQUIRE(VALID_RESPONSE(resp)); 511 REQUIRE(VALID_DISPATCH(resp->disp)); 512 513 disp = resp->disp; 514 515 REQUIRE(disp->tid == isc_tid()); 516 INSIST(resp->reading); 517 resp->reading = false; 518 519 if (resp->state == DNS_DISPATCHSTATE_CANCELED) { 520 /* 521 * Nobody is interested in the callback if the response 522 * has been canceled already. Detach from the response 523 * and the handle. 524 */ 525 respond = false; 526 eresult = ISC_R_CANCELED; 527 } 528 529 dispentry_log(resp, ISC_LOG_DEBUG(90), 530 "read callback:%s, requests %" PRIuFAST32, 531 isc_result_totext(eresult), disp->requests); 532 533 if (eresult != ISC_R_SUCCESS) { 534 /* 535 * This is most likely a network error on a connected 536 * socket, a timeout, or the query has been canceled. 537 * It makes no sense to check the address or parse the 538 * packet, but we can return the error to the caller. 539 */ 540 goto done; 541 } 542 543 peer = isc_nmhandle_peeraddr(handle); 544 isc_netaddr_fromsockaddr(&netaddr, &peer); 545 546 /* 547 * If this is from a blackholed address, drop it. 548 */ 549 if (disp->mgr->blackhole != NULL && 550 dns_acl_match(&netaddr, NULL, disp->mgr->blackhole, NULL, &match, 551 NULL) == ISC_R_SUCCESS && 552 match > 0) 553 { 554 if (isc_log_wouldlog(dns_lctx, ISC_LOG_DEBUG(10))) { 555 char netaddrstr[ISC_NETADDR_FORMATSIZE]; 556 isc_netaddr_format(&netaddr, netaddrstr, 557 sizeof(netaddrstr)); 558 dispentry_log(resp, ISC_LOG_DEBUG(10), 559 "blackholed packet from %s", netaddrstr); 560 } 561 goto next; 562 } 563 564 /* 565 * Peek into the buffer to see what we can see. 566 */ 567 id = resp->id; 568 isc_buffer_init(&source, region->base, region->length); 569 isc_buffer_add(&source, region->length); 570 dres = dns_message_peekheader(&source, &id, &flags); 571 if (dres != ISC_R_SUCCESS) { 572 char netaddrstr[ISC_NETADDR_FORMATSIZE]; 573 isc_netaddr_format(&netaddr, netaddrstr, sizeof(netaddrstr)); 574 dispentry_log(resp, ISC_LOG_DEBUG(10), 575 "got garbage packet from %s", netaddrstr); 576 goto next; 577 } 578 579 dispentry_log(resp, ISC_LOG_DEBUG(92), 580 "got valid DNS message header, /QR %c, id %u", 581 ((flags & DNS_MESSAGEFLAG_QR) != 0) ? '1' : '0', id); 582 583 /* 584 * Look at the message flags. If it's a query, ignore it. 585 */ 586 if ((flags & DNS_MESSAGEFLAG_QR) == 0) { 587 goto next; 588 } 589 590 /* 591 * The QID and the address must match the expected ones. A 592 * mismatch can happen during normal operation only when a stale 593 * response from a previous query arrives late, which is rare in 594 * practice; treat any mismatch as a possible spoofing attempt and 595 * let the caller retry over TCP to prevent off-path spoofing. 596 */ 597 if (resp->id != id || !isc_sockaddr_equal(&peer, &resp->peer)) { 598 dispentry_log(resp, ISC_LOG_DEBUG(90), 599 "response doesn't match"); 600 inc_stats(disp->mgr, dns_resstatscounter_mismatch); 601 eresult = DNS_R_MISMATCH; 602 goto done; 603 } 604 605 /* 606 * We have the right resp, so call the caller back. 607 */ 608 goto done; 609 610 next: 611 /* 612 * This is the wrong response. Check whether there is still enough 613 * time to wait for the correct one to arrive before the timeout fires. 614 */ 615 now = isc_loop_now(resp->loop); 616 if (resp->timeout > 0) { 617 timeout = resp->timeout - dispentry_runtime(resp, &now); 618 if (timeout <= 0) { 619 /* 620 * The time window for receiving the correct response is 621 * already closed, libuv has just not processed the 622 * socket timer yet. Invoke the read callback, 623 * indicating a timeout. 624 */ 625 eresult = ISC_R_TIMEDOUT; 626 goto done; 627 } 628 } 629 630 /* 631 * Do not invoke the read callback just yet and instead wait for the 632 * proper response to arrive until the original timeout fires. 633 */ 634 respond = false; 635 udp_dispatch_getnext(resp, timeout); 636 637 done: 638 if (respond) { 639 dispentry_log(resp, ISC_LOG_DEBUG(90), 640 "UDP read callback on %p: %s", handle, 641 isc_result_totext(eresult)); 642 resp->response(eresult, region, resp->arg); 643 } 644 645 dns_dispentry_detach(&resp); /* DISPENTRY003 */ 646 } 647 648 static isc_result_t 649 tcp_recv_oldest(dns_dispatch_t *disp, dns_dispentry_t **respp) { 650 dns_dispentry_t *resp = NULL; 651 resp = ISC_LIST_HEAD(disp->active); 652 if (resp != NULL) { 653 disp->timedout++; 654 655 *respp = resp; 656 return ISC_R_TIMEDOUT; 657 } 658 659 return ISC_R_NOTFOUND; 660 } 661 662 static isc_result_t 663 tcp_recv_success(dns_dispatch_t *disp, isc_region_t *region, 664 dns_dispentry_t **respp) { 665 isc_buffer_t source; 666 dns_messageid_t id; 667 unsigned int flags; 668 isc_result_t result = ISC_R_SUCCESS; 669 670 dispatch_log(disp, ISC_LOG_DEBUG(90), 671 "TCP read success, length == %d, addr = %p", 672 region->length, region->base); 673 674 /* 675 * Peek into the buffer to see what we can see. 676 */ 677 isc_buffer_init(&source, region->base, region->length); 678 isc_buffer_add(&source, region->length); 679 result = dns_message_peekheader(&source, &id, &flags); 680 if (result != ISC_R_SUCCESS) { 681 dispatch_log(disp, ISC_LOG_DEBUG(10), "got garbage packet"); 682 return ISC_R_UNEXPECTED; 683 } 684 685 dispatch_log(disp, ISC_LOG_DEBUG(92), 686 "got valid DNS message header, /QR %c, id %u", 687 ((flags & DNS_MESSAGEFLAG_QR) != 0) ? '1' : '0', id); 688 689 /* 690 * Look at the message flags. If it's a query, ignore it and keep 691 * reading. 692 */ 693 if ((flags & DNS_MESSAGEFLAG_QR) == 0) { 694 dispatch_log(disp, ISC_LOG_DEBUG(10), 695 "got DNS query instead of answer"); 696 return ISC_R_UNEXPECTED; 697 } 698 699 /* 700 * We have a valid response; find the associated dispentry by 701 * scanning disp->active. With sequential IDs and a bounded 702 * pipelining limit this is a short linear scan. 703 */ 704 dns_dispentry_t *resp = NULL, *r = NULL; 705 ISC_LIST_FOREACH(disp->active, r, alink) { 706 if (r->id == id) { 707 resp = r; 708 break; 709 } 710 } 711 712 if (resp != NULL) { 713 *respp = resp; 714 } else { 715 result = ISC_R_NOTFOUND; 716 } 717 dispatch_log(disp, ISC_LOG_DEBUG(90), "search for response: %s", 718 isc_result_totext(result)); 719 720 return result; 721 } 722 723 static void 724 tcp_recv_add(dns_displist_t *resps, dns_dispentry_t *resp, 725 isc_result_t result) { 726 dns_dispentry_ref(resp); /* DISPENTRY009 */ 727 ISC_LIST_UNLINK(resp->disp->active, resp, alink); 728 ISC_LIST_APPEND(*resps, resp, rlink); 729 INSIST(resp->reading); 730 resp->reading = false; 731 resp->result = result; 732 } 733 734 static void 735 tcp_recv_shutdown(dns_dispatch_t *disp, dns_displist_t *resps, 736 isc_result_t result) { 737 dns_dispentry_t *resp = NULL, *next = NULL; 738 739 /* 740 * If there are any active responses, shut them all down. 741 */ 742 for (resp = ISC_LIST_HEAD(disp->active); resp != NULL; resp = next) { 743 next = ISC_LIST_NEXT(resp, alink); 744 tcp_recv_add(resps, resp, result); 745 } 746 disp->state = DNS_DISPATCHSTATE_CANCELED; 747 748 /* Close now so the socket doesn't linger in CLOSE-WAIT. */ 749 if (disp->handle != NULL) { 750 isc_nmhandle_close(disp->handle); 751 } 752 } 753 754 static void 755 tcp_recv_processall(dns_displist_t *resps, isc_region_t *region) { 756 dns_dispentry_t *resp = NULL, *next = NULL; 757 758 for (resp = ISC_LIST_HEAD(*resps); resp != NULL; resp = next) { 759 next = ISC_LIST_NEXT(resp, rlink); 760 ISC_LIST_UNLINK(*resps, resp, rlink); 761 762 dispentry_log(resp, ISC_LOG_DEBUG(90), "read callback: %s", 763 isc_result_totext(resp->result)); 764 resp->response(resp->result, region, resp->arg); 765 dns_dispentry_detach(&resp); /* DISPENTRY009 */ 766 } 767 } 768 769 static void 770 tcp_startrecv_idle(dns_dispatch_t *disp) { 771 REQUIRE(disp->mgr->reuse_timeout > 0); 772 773 /* 774 * No outstanding responses, but the connection is still up and 775 * reusable. Keep a read pending (bounded by the idle timeout) 776 * so that a peer-initiated close is detected promptly; 777 * otherwise the socket would linger in CLOSE-WAIT and be handed 778 * out again as a dead reused connection. 779 */ 780 isc_nmhandle_cleartimeout(disp->handle); 781 isc_nmhandle_settimeout(disp->handle, disp->mgr->reuse_timeout); 782 if (!disp->reading) { 783 dispatch_log(disp, ISC_LOG_DEBUG(90), "keeping idle read on %p", 784 disp->handle); 785 tcp_startrecv(disp, NULL); 786 } 787 } 788 789 /* 790 * General flow: 791 * 792 * If I/O result == CANCELED, EOF, or error, notify everyone as the 793 * various queues drain. 794 * 795 * If response: 796 * Allocate event, fill in details. 797 * If cannot allocate, restart. 798 * find target. If not found, restart. 799 * if event queue is not empty, queue. else, send. 800 * restart. 801 */ 802 static void 803 tcp_recv(isc_nmhandle_t *handle, isc_result_t eresult, isc_region_t *region, 804 void *arg) { 805 dns_dispatch_t *disp = (dns_dispatch_t *)arg; 806 dns_dispentry_t *resp = NULL; 807 char buf[ISC_SOCKADDR_FORMATSIZE]; 808 isc_sockaddr_t peer; 809 dns_displist_t resps = ISC_LIST_INITIALIZER; 810 isc_time_t now; 811 int timeout = 0; 812 isc_result_t result = eresult; 813 814 REQUIRE(VALID_DISPATCH(disp)); 815 816 REQUIRE(disp->tid == isc_tid()); 817 INSIST(disp->reading); 818 disp->reading = false; 819 820 dispatch_log(disp, ISC_LOG_DEBUG(90), 821 "TCP read:%s:requests %" PRIuFAST32, 822 isc_result_totext(eresult), disp->requests); 823 824 peer = isc_nmhandle_peeraddr(handle); 825 826 rcu_read_lock(); 827 /* 828 * Phase 1: Process timeout and success. 829 */ 830 switch (eresult) { 831 case ISC_R_TIMEDOUT: 832 /* 833 * Time out the oldest response in the active queue. 834 */ 835 result = tcp_recv_oldest(disp, &resp); 836 break; 837 case ISC_R_SUCCESS: 838 /* We got an answer */ 839 result = tcp_recv_success(disp, region, &resp); 840 break; 841 842 default: 843 break; 844 } 845 846 if (resp != NULL) { 847 tcp_recv_add(&resps, resp, result); 848 } 849 850 /* 851 * Phase 2: Look if we timed out before. 852 */ 853 854 if (result == ISC_R_NOTFOUND) { 855 if (eresult == ISC_R_TIMEDOUT) { 856 /* 857 * An idle keep-alive read timed out with no outstanding 858 * responses. 859 */ 860 result = ISC_R_CANCELED; 861 } else if (disp->timedout > 0) { 862 /* There was active query that timed-out before */ 863 disp->timedout--; 864 } else { 865 result = ISC_R_UNEXPECTED; 866 } 867 } 868 869 /* 870 * Phase 3: Trigger timeouts. It's possible that the responses would 871 * have been timed out out already, but non-matching TCP reads have 872 * prevented this. 873 */ 874 resp = ISC_LIST_HEAD(disp->active); 875 if (resp != NULL) { 876 now = isc_loop_now(resp->loop); 877 } 878 while (resp != NULL) { 879 dns_dispentry_t *next = ISC_LIST_NEXT(resp, alink); 880 881 if (resp->timeout > 0) { 882 timeout = resp->timeout - dispentry_runtime(resp, &now); 883 if (timeout <= 0) { 884 tcp_recv_add(&resps, resp, ISC_R_TIMEDOUT); 885 } 886 } 887 888 resp = next; 889 } 890 891 /* 892 * Phase 4: log if we errored out. 893 */ 894 switch (result) { 895 case ISC_R_SUCCESS: 896 case ISC_R_TIMEDOUT: 897 case ISC_R_NOTFOUND: 898 break; 899 case ISC_R_SHUTTINGDOWN: 900 case ISC_R_CANCELED: 901 case ISC_R_EOF: 902 case ISC_R_CONNECTIONRESET: 903 if (isc_log_wouldlog(dns_lctx, ISC_LOG_DEBUG(90))) { 904 isc_sockaddr_format(&peer, buf, sizeof(buf)); 905 dispatch_log(disp, ISC_LOG_DEBUG(90), 906 "shutting down TCP: %s: %s", buf, 907 isc_result_totext(result)); 908 } 909 tcp_recv_shutdown(disp, &resps, result); 910 break; 911 default: 912 isc_sockaddr_format(&peer, buf, sizeof(buf)); 913 dispatch_log(disp, ISC_LOG_ERROR, 914 "shutting down due to TCP " 915 "receive error: %s: %s", 916 buf, isc_result_totext(result)); 917 tcp_recv_shutdown(disp, &resps, result); 918 break; 919 } 920 921 /* 922 * Phase 5: Resume reading if there are still active responses 923 */ 924 resp = ISC_LIST_HEAD(disp->active); 925 if (resp != NULL) { 926 if (resp->timeout > 0) { 927 timeout = resp->timeout - dispentry_runtime(resp, &now); 928 INSIST(timeout > 0); 929 } 930 tcp_startrecv(disp, resp); 931 if (timeout > 0) { 932 isc_nmhandle_settimeout(handle, timeout); 933 } 934 } 935 936 rcu_read_unlock(); 937 938 /* 939 * Phase 6: Process all scheduled callbacks. 940 */ 941 tcp_recv_processall(&resps, region); 942 943 /* 944 * Phase 7: Restart reading on idle connections. 945 */ 946 if (ISC_LIST_EMPTY(disp->active) && 947 disp->state == DNS_DISPATCHSTATE_CONNECTED && 948 disp->mgr->reuse_timeout > 0) 949 { 950 tcp_startrecv_idle(disp); 951 } 952 953 dns_dispatch_detach(&disp); /* DISPATCH002 */ 954 } 955 956 /*% 957 * Create a temporary port list to set the initial default set of dispatch 958 * ephemeral ports. This is almost meaningless as the application will 959 * normally set the ports explicitly, but is provided to fill some minor corner 960 * cases. 961 */ 962 static void 963 create_default_portset(isc_mem_t *mctx, int family, isc_portset_t **portsetp) { 964 in_port_t low, high; 965 966 isc_net_getportrange(family, &low, &high); 967 968 isc_portset_create(mctx, portsetp); 969 isc_portset_addrange(*portsetp, low, high); 970 } 971 972 static isc_result_t 973 setavailports(dns_dispatchmgr_t *mgr, isc_portset_t *v4portset, 974 isc_portset_t *v6portset) { 975 in_port_t *v4ports, *v6ports, p = 0; 976 unsigned int nv4ports, nv6ports, i4 = 0, i6 = 0; 977 978 nv4ports = isc_portset_nports(v4portset); 979 nv6ports = isc_portset_nports(v6portset); 980 981 v4ports = NULL; 982 if (nv4ports != 0) { 983 v4ports = isc_mem_cget(mgr->mctx, nv4ports, sizeof(in_port_t)); 984 } 985 v6ports = NULL; 986 if (nv6ports != 0) { 987 v6ports = isc_mem_cget(mgr->mctx, nv6ports, sizeof(in_port_t)); 988 } 989 990 do { 991 if (isc_portset_isset(v4portset, p)) { 992 INSIST(i4 < nv4ports); 993 v4ports[i4++] = p; 994 } 995 if (isc_portset_isset(v6portset, p)) { 996 INSIST(i6 < nv6ports); 997 v6ports[i6++] = p; 998 } 999 } while (p++ < 65535); 1000 INSIST(i4 == nv4ports && i6 == nv6ports); 1001 1002 if (mgr->v4ports != NULL) { 1003 isc_mem_cput(mgr->mctx, mgr->v4ports, mgr->nv4ports, 1004 sizeof(in_port_t)); 1005 } 1006 mgr->v4ports = v4ports; 1007 mgr->nv4ports = nv4ports; 1008 1009 if (mgr->v6ports != NULL) { 1010 isc_mem_cput(mgr->mctx, mgr->v6ports, mgr->nv6ports, 1011 sizeof(in_port_t)); 1012 } 1013 mgr->v6ports = v6ports; 1014 mgr->nv6ports = nv6ports; 1015 1016 return ISC_R_SUCCESS; 1017 } 1018 1019 /* 1020 * Publics. 1021 */ 1022 1023 isc_result_t 1024 dns_dispatchmgr_create(isc_mem_t *mctx, isc_loopmgr_t *loopmgr, isc_nm_t *nm, 1025 dns_dispatchmgr_t **mgrp) { 1026 dns_dispatchmgr_t *mgr = NULL; 1027 isc_portset_t *v4portset = NULL; 1028 isc_portset_t *v6portset = NULL; 1029 1030 REQUIRE(mctx != NULL); 1031 REQUIRE(mgrp != NULL && *mgrp == NULL); 1032 1033 mgr = isc_mem_get(mctx, sizeof(dns_dispatchmgr_t)); 1034 *mgr = (dns_dispatchmgr_t){ 1035 .magic = 0, 1036 .nloops = isc_loopmgr_nloops(loopmgr), 1037 }; 1038 1039 #if DNS_DISPATCH_TRACE 1040 fprintf(stderr, "dns_dispatchmgr__init:%s:%s:%d:%p->references = 1\n", 1041 __func__, __FILE__, __LINE__, mgr); 1042 #endif 1043 isc_refcount_init(&mgr->references, 1); 1044 1045 isc_mem_attach(mctx, &mgr->mctx); 1046 isc_nm_attach(nm, &mgr->nm); 1047 1048 mgr->tcps = isc_mem_cget(mgr->mctx, mgr->nloops, sizeof(mgr->tcps[0])); 1049 for (size_t i = 0; i < mgr->nloops; i++) { 1050 mgr->tcps[i] = cds_lfht_new( 1051 2, 2, 0, CDS_LFHT_AUTO_RESIZE | CDS_LFHT_ACCOUNTING, 1052 NULL); 1053 } 1054 1055 create_default_portset(mgr->mctx, AF_INET, &v4portset); 1056 create_default_portset(mgr->mctx, AF_INET6, &v6portset); 1057 1058 setavailports(mgr, v4portset, v6portset); 1059 1060 isc_portset_destroy(mgr->mctx, &v4portset); 1061 isc_portset_destroy(mgr->mctx, &v6portset); 1062 1063 mgr->qids = cds_lfht_new(QIDS_INIT_SIZE, QIDS_MIN_SIZE, 0, 1064 CDS_LFHT_AUTO_RESIZE | CDS_LFHT_ACCOUNTING, 1065 NULL); 1066 1067 mgr->magic = DNS_DISPATCHMGR_MAGIC; 1068 1069 *mgrp = mgr; 1070 return ISC_R_SUCCESS; 1071 } 1072 1073 #if DNS_DISPATCH_TRACE 1074 ISC_REFCOUNT_TRACE_IMPL(dns_dispatchmgr, dispatchmgr_destroy); 1075 #else 1076 ISC_REFCOUNT_IMPL(dns_dispatchmgr, dispatchmgr_destroy); 1077 #endif 1078 1079 void 1080 dns_dispatchmgr_setblackhole(dns_dispatchmgr_t *mgr, dns_acl_t *blackhole) { 1081 REQUIRE(VALID_DISPATCHMGR(mgr)); 1082 if (mgr->blackhole != NULL) { 1083 dns_acl_detach(&mgr->blackhole); 1084 } 1085 dns_acl_attach(blackhole, &mgr->blackhole); 1086 } 1087 1088 dns_acl_t * 1089 dns_dispatchmgr_getblackhole(dns_dispatchmgr_t *mgr) { 1090 REQUIRE(VALID_DISPATCHMGR(mgr)); 1091 return mgr->blackhole; 1092 } 1093 1094 isc_result_t 1095 dns_dispatchmgr_setavailports(dns_dispatchmgr_t *mgr, isc_portset_t *v4portset, 1096 isc_portset_t *v6portset) { 1097 REQUIRE(VALID_DISPATCHMGR(mgr)); 1098 return setavailports(mgr, v4portset, v6portset); 1099 } 1100 1101 void 1102 dns_dispatchmgr_setreusetimeout(dns_dispatchmgr_t *mgr, 1103 unsigned int reuse_timeout) { 1104 REQUIRE(VALID_DISPATCHMGR(mgr)); 1105 mgr->reuse_timeout = reuse_timeout; 1106 } 1107 1108 static void 1109 dispatchmgr_destroy(dns_dispatchmgr_t *mgr) { 1110 REQUIRE(VALID_DISPATCHMGR(mgr)); 1111 1112 isc_refcount_destroy(&mgr->references); 1113 1114 mgr->magic = 0; 1115 1116 RUNTIME_CHECK(!cds_lfht_destroy(mgr->qids, NULL)); 1117 1118 for (size_t i = 0; i < mgr->nloops; i++) { 1119 RUNTIME_CHECK(!cds_lfht_destroy(mgr->tcps[i], NULL)); 1120 } 1121 isc_mem_cput(mgr->mctx, mgr->tcps, mgr->nloops, sizeof(mgr->tcps[0])); 1122 1123 if (mgr->blackhole != NULL) { 1124 dns_acl_detach(&mgr->blackhole); 1125 } 1126 1127 if (mgr->stats != NULL) { 1128 isc_stats_detach(&mgr->stats); 1129 } 1130 1131 if (mgr->v4ports != NULL) { 1132 isc_mem_cput(mgr->mctx, mgr->v4ports, mgr->nv4ports, 1133 sizeof(in_port_t)); 1134 } 1135 if (mgr->v6ports != NULL) { 1136 isc_mem_cput(mgr->mctx, mgr->v6ports, mgr->nv6ports, 1137 sizeof(in_port_t)); 1138 } 1139 1140 isc_nm_detach(&mgr->nm); 1141 1142 isc_mem_putanddetach(&mgr->mctx, mgr, sizeof(dns_dispatchmgr_t)); 1143 } 1144 1145 void 1146 dns_dispatchmgr_setstats(dns_dispatchmgr_t *mgr, isc_stats_t *stats) { 1147 REQUIRE(VALID_DISPATCHMGR(mgr)); 1148 REQUIRE(mgr->stats == NULL); 1149 1150 isc_stats_attach(stats, &mgr->stats); 1151 } 1152 1153 /* 1154 * Allocate and set important limits. 1155 */ 1156 static void 1157 dispatch_allocate(dns_dispatchmgr_t *mgr, isc_socktype_t type, uint32_t tid, 1158 dns_dispatch_t **dispp) { 1159 dns_dispatch_t *disp = NULL; 1160 1161 REQUIRE(VALID_DISPATCHMGR(mgr)); 1162 REQUIRE(dispp != NULL && *dispp == NULL); 1163 1164 /* 1165 * Set up the dispatcher, mostly. Don't bother setting some of 1166 * the options that are controlled by tcp vs. udp, etc. 1167 */ 1168 1169 disp = isc_mem_get(mgr->mctx, sizeof(*disp)); 1170 *disp = (dns_dispatch_t){ 1171 .socktype = type, 1172 .active = ISC_LIST_INITIALIZER, 1173 .pending = ISC_LIST_INITIALIZER, 1174 .tid = tid, 1175 .magic = DISPATCH_MAGIC, 1176 }; 1177 1178 isc_mem_attach(mgr->mctx, &disp->mctx); 1179 1180 dns_dispatchmgr_attach(mgr, &disp->mgr); 1181 #if DNS_DISPATCH_TRACE 1182 fprintf(stderr, "dns_dispatch__init:%s:%s:%d:%p->references = 1\n", 1183 __func__, __FILE__, __LINE__, disp); 1184 #endif 1185 isc_refcount_init(&disp->references, 1); /* DISPATCH000 */ 1186 1187 *dispp = disp; 1188 } 1189 1190 struct dispatch_key { 1191 const isc_sockaddr_t *local; 1192 const isc_sockaddr_t *peer; 1193 const dns_transport_t *transport; 1194 const dns_dispatchtype_t disptype; 1195 }; 1196 1197 static uint32_t 1198 dispatch_hash(struct dispatch_key *key) { 1199 isc_hash32_t hash; 1200 1201 isc_hash32_init(&hash); 1202 1203 isc_sockaddr_hash_ex(&hash, key->peer, false); 1204 if (key->local != NULL) { 1205 isc_sockaddr_hash_ex(&hash, key->local, true); 1206 } 1207 if (key->transport != NULL) { 1208 uintptr_t transport = (uintptr_t)key->transport; 1209 isc_hash32_hash(&hash, &transport, sizeof(transport), true); 1210 } 1211 isc_hash32_hash(&hash, &key->disptype, sizeof(key->disptype), true); 1212 1213 return isc_hash32_finalize(&hash); 1214 } 1215 1216 static int 1217 dispatch_match(struct cds_lfht_node *node, const void *key0) { 1218 dns_dispatch_t *disp = caa_container_of(node, dns_dispatch_t, ht_node); 1219 const struct dispatch_key *key = key0; 1220 1221 return disp->transport == key->transport && 1222 isc_sockaddr_equal(&disp->peer, key->peer) && 1223 (key->local == NULL || 1224 isc_sockaddr_equal(&disp->local, key->local)); 1225 } 1226 1227 static isc_result_t 1228 dispatch_gettcp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 1229 const isc_sockaddr_t *destaddr, dns_transport_t *transport, 1230 dns_dispatchtype_t disptype, dns_dispatch_t **dispp) { 1231 dns_dispatch_t *disp_connected = NULL; 1232 dns_dispatch_t *disp_fallback = NULL; 1233 isc_result_t result = ISC_R_NOTFOUND; 1234 uint32_t tid = isc_tid(); 1235 1236 REQUIRE(VALID_DISPATCHMGR(mgr)); 1237 REQUIRE(destaddr != NULL); 1238 REQUIRE(dispp != NULL && *dispp == NULL); 1239 1240 struct dispatch_key key = { 1241 .local = localaddr, 1242 .peer = destaddr, 1243 .transport = transport, 1244 .disptype = disptype, 1245 }; 1246 1247 rcu_read_lock(); 1248 struct cds_lfht_iter iter; 1249 dns_dispatch_t *disp = NULL; 1250 cds_lfht_for_each_entry_duplicate(mgr->tcps[tid], dispatch_hash(&key), 1251 dispatch_match, &key, &iter, disp, 1252 ht_node) { 1253 INSIST(disp->tid == isc_tid()); 1254 INSIST(disp->socktype == isc_socktype_tcp); 1255 1256 switch (disp->state) { 1257 case DNS_DISPATCHSTATE_NONE: 1258 /* A dispatch in indeterminate state, skip it */ 1259 break; 1260 case DNS_DISPATCHSTATE_CONNECTED: 1261 /* We found a connected dispatch */ 1262 dns_dispatch_attach(disp, &disp_connected); 1263 break; 1264 case DNS_DISPATCHSTATE_CONNECTING: 1265 /* We found "a" dispatch, store it for later */ 1266 if (disp_fallback == NULL) { 1267 dns_dispatch_attach(disp, &disp_fallback); 1268 } 1269 break; 1270 case DNS_DISPATCHSTATE_CANCELED: 1271 /* A canceled dispatch, skip it. */ 1272 break; 1273 default: 1274 UNREACHABLE(); 1275 } 1276 1277 if (disp_connected != NULL) { 1278 break; 1279 } 1280 } 1281 rcu_read_unlock(); 1282 1283 if (disp_connected != NULL) { 1284 /* We found connected dispatch */ 1285 INSIST(disp_connected->handle != NULL); 1286 1287 *dispp = disp_connected; 1288 disp_connected = NULL; 1289 1290 result = ISC_R_SUCCESS; 1291 1292 if (disp_fallback != NULL) { 1293 dns_dispatch_detach(&disp_fallback); 1294 } 1295 } else if (disp_fallback != NULL) { 1296 *dispp = disp_fallback; 1297 1298 result = ISC_R_SUCCESS; 1299 } 1300 1301 return result; 1302 } 1303 1304 static void 1305 dispatch_createtcp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 1306 const isc_sockaddr_t *destaddr, dns_transport_t *transport, 1307 dns_dispatchtype_t disptype, dns_dispatchopt_t options, 1308 dns_dispatch_t **dispp) { 1309 dns_dispatch_t *disp = NULL; 1310 uint32_t tid = isc_tid(); 1311 1312 dispatch_allocate(mgr, isc_socktype_tcp, tid, &disp); 1313 1314 disp->disptype = disptype; 1315 disp->nextid = isc_random16(); 1316 disp->options = options; 1317 disp->peer = *destaddr; 1318 if (transport != NULL) { 1319 dns_transport_attach(transport, &disp->transport); 1320 } 1321 1322 if (localaddr != NULL) { 1323 disp->local = *localaddr; 1324 } else { 1325 int pf; 1326 pf = isc_sockaddr_pf(destaddr); 1327 isc_sockaddr_anyofpf(&disp->local, pf); 1328 isc_sockaddr_setport(&disp->local, 0); 1329 } 1330 1331 /* 1332 * Append it to the dispatcher list. 1333 */ 1334 if ((options & DNS_DISPATCHOPT_FIXEDID) == 0) { 1335 struct dispatch_key key = { 1336 .local = &disp->local, 1337 .peer = &disp->peer, 1338 .transport = transport, 1339 .disptype = disptype, 1340 }; 1341 rcu_read_lock(); 1342 cds_lfht_add(mgr->tcps[tid], dispatch_hash(&key), 1343 &disp->ht_node); 1344 rcu_read_unlock(); 1345 } 1346 1347 *dispp = disp; 1348 } 1349 1350 isc_result_t 1351 dns_dispatch_createtcp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 1352 const isc_sockaddr_t *destaddr, 1353 dns_transport_t *transport, dns_dispatchtype_t disptype, 1354 dns_dispatchopt_t options, dns_dispatch_t **dispp) { 1355 REQUIRE(VALID_DISPATCHMGR(mgr)); 1356 REQUIRE(destaddr != NULL); 1357 1358 isc_result_t result; 1359 1360 if ((options & DNS_DISPATCHOPT_FIXEDID) == 0 && 1361 disptype != DNS_DISPATCHTYPE_XFRIN) 1362 { 1363 result = dispatch_gettcp(mgr, localaddr, destaddr, transport, 1364 disptype, dispp); 1365 if (result == ISC_R_SUCCESS) { 1366 if (isc_log_wouldlog(dns_lctx, 90)) { 1367 char addrbuf[ISC_SOCKADDR_FORMATSIZE]; 1368 1369 isc_sockaddr_format(&(*dispp)->local, addrbuf, 1370 ISC_SOCKADDR_FORMATSIZE); 1371 1372 mgr_log(mgr, ISC_LOG_DEBUG(90), 1373 "dns_dispatch_createtcp: reused TCP " 1374 "dispatch %p for " 1375 "%s", 1376 *dispp, addrbuf); 1377 } 1378 return result; 1379 } 1380 } 1381 1382 /* 1383 * Otherwise allocate new TCP dispatch. 1384 */ 1385 1386 dispatch_createtcp(mgr, localaddr, destaddr, transport, disptype, 1387 options, dispp); 1388 1389 if (isc_log_wouldlog(dns_lctx, 90)) { 1390 char addrbuf[ISC_SOCKADDR_FORMATSIZE]; 1391 1392 isc_sockaddr_format(&(*dispp)->local, addrbuf, 1393 ISC_SOCKADDR_FORMATSIZE); 1394 1395 mgr_log(mgr, ISC_LOG_DEBUG(90), 1396 "dns_dispatch_createtcp: created TCP dispatch %p for " 1397 "%s", 1398 *dispp, addrbuf); 1399 } 1400 1401 return ISC_R_SUCCESS; 1402 } 1403 1404 isc_result_t 1405 dns_dispatch_createudp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 1406 dns_dispatch_t **dispp) { 1407 isc_result_t result; 1408 dns_dispatch_t *disp = NULL; 1409 1410 REQUIRE(VALID_DISPATCHMGR(mgr)); 1411 REQUIRE(localaddr != NULL); 1412 REQUIRE(dispp != NULL && *dispp == NULL); 1413 1414 result = dispatch_createudp(mgr, localaddr, isc_tid(), &disp); 1415 if (result == ISC_R_SUCCESS) { 1416 *dispp = disp; 1417 } 1418 1419 return result; 1420 } 1421 1422 static isc_result_t 1423 dispatch_createudp(dns_dispatchmgr_t *mgr, const isc_sockaddr_t *localaddr, 1424 uint32_t tid, dns_dispatch_t **dispp) { 1425 isc_result_t result = ISC_R_SUCCESS; 1426 dns_dispatch_t *disp = NULL; 1427 isc_sockaddr_t sa_any; 1428 1429 /* 1430 * Check whether this address/port is available locally. 1431 */ 1432 isc_sockaddr_anyofpf(&sa_any, isc_sockaddr_pf(localaddr)); 1433 if (!isc_sockaddr_eqaddr(&sa_any, localaddr)) { 1434 result = isc_nm_checkaddr(localaddr, isc_socktype_udp); 1435 if (result != ISC_R_SUCCESS) { 1436 return result; 1437 } 1438 } 1439 1440 dispatch_allocate(mgr, isc_socktype_udp, tid, &disp); 1441 1442 if (isc_log_wouldlog(dns_lctx, 90)) { 1443 char addrbuf[ISC_SOCKADDR_FORMATSIZE]; 1444 1445 isc_sockaddr_format(localaddr, addrbuf, 1446 ISC_SOCKADDR_FORMATSIZE); 1447 mgr_log(mgr, ISC_LOG_DEBUG(90), 1448 "dispatch_createudp: created UDP dispatch %p for %s", 1449 disp, addrbuf); 1450 } 1451 1452 disp->local = *localaddr; 1453 1454 /* 1455 * Don't append it to the dispatcher list, we don't care about UDP, only 1456 * TCP should be searched 1457 * 1458 * ISC_LIST_APPEND(mgr->list, disp, link); 1459 */ 1460 1461 *dispp = disp; 1462 1463 return result; 1464 } 1465 1466 static void 1467 dispatch_destroy_rcu(struct rcu_head *rcu_head) { 1468 dns_dispatch_t *disp = caa_container_of(rcu_head, dns_dispatch_t, 1469 rcu_head); 1470 1471 isc_mem_putanddetach(&disp->mctx, disp, sizeof(*disp)); 1472 } 1473 1474 static void 1475 dispatch_destroy(dns_dispatch_t *disp) { 1476 dns_dispatchmgr_t *mgr = disp->mgr; 1477 uint32_t tid = isc_tid(); 1478 1479 disp->magic = 0; 1480 1481 if ((disp->options & DNS_DISPATCHOPT_FIXEDID) == 0 && 1482 disp->socktype == isc_socktype_tcp) 1483 { 1484 (void)cds_lfht_del(mgr->tcps[tid], &disp->ht_node); 1485 } 1486 1487 INSIST(disp->requests == 0); 1488 INSIST(ISC_LIST_EMPTY(disp->pending)); 1489 INSIST(ISC_LIST_EMPTY(disp->active)); 1490 1491 dispatch_log(disp, ISC_LOG_DEBUG(90), "destroying dispatch %p", disp); 1492 1493 if (disp->handle) { 1494 dispatch_log(disp, ISC_LOG_DEBUG(90), 1495 "detaching TCP handle %p from %p", disp->handle, 1496 &disp->handle); 1497 isc_nmhandle_detach(&disp->handle); 1498 } 1499 if (disp->transport != NULL) { 1500 dns_transport_detach(&disp->transport); 1501 } 1502 dns_dispatchmgr_detach(&disp->mgr); 1503 1504 call_rcu(&disp->rcu_head, dispatch_destroy_rcu); 1505 } 1506 1507 #if DNS_DISPATCH_TRACE 1508 ISC_REFCOUNT_TRACE_IMPL(dns_dispatch, dispatch_destroy); 1509 #else 1510 ISC_REFCOUNT_IMPL(dns_dispatch, dispatch_destroy); 1511 #endif 1512 1513 isc_result_t 1514 dns_dispatch_add(dns_dispatch_t *disp, isc_loop_t *loop, 1515 dns_dispatchopt_t options, unsigned int timeout, 1516 const isc_sockaddr_t *dest, dns_transport_t *transport, 1517 isc_tlsctx_cache_t *tlsctx_cache, dispatch_cb_t connected, 1518 dispatch_cb_t sent, dispatch_cb_t response, void *arg, 1519 dns_messageid_t *idp, dns_dispentry_t **respp) { 1520 REQUIRE(VALID_DISPATCH(disp)); 1521 REQUIRE(dest != NULL); 1522 REQUIRE(respp != NULL && *respp == NULL); 1523 REQUIRE(idp != NULL); 1524 REQUIRE(disp->socktype == isc_socktype_tcp || 1525 disp->socktype == isc_socktype_udp); 1526 REQUIRE(connected != NULL); 1527 REQUIRE(response != NULL); 1528 REQUIRE(sent != NULL); 1529 REQUIRE(loop != NULL); 1530 REQUIRE(disp->tid == isc_tid()); 1531 REQUIRE(disp->transport == transport); 1532 1533 if (disp->state == DNS_DISPATCHSTATE_CANCELED) { 1534 return ISC_R_CANCELED; 1535 } 1536 1537 in_port_t localport = isc_sockaddr_getport(&disp->local); 1538 dns_dispentry_t *resp = isc_mem_get(disp->mctx, sizeof(*resp)); 1539 *resp = (dns_dispentry_t){ 1540 .timeout = timeout, 1541 .port = localport, 1542 .peer = *dest, 1543 .loop = loop, 1544 .connected = connected, 1545 .sent = sent, 1546 .response = response, 1547 .arg = arg, 1548 .alink = ISC_LINK_INITIALIZER, 1549 .plink = ISC_LINK_INITIALIZER, 1550 .rlink = ISC_LINK_INITIALIZER, 1551 .magic = RESPONSE_MAGIC, 1552 }; 1553 1554 #if DNS_DISPATCH_TRACE 1555 fprintf(stderr, "dns_dispentry__init:%s:%s:%d:%p->references = 1\n", 1556 __func__, __FILE__, __LINE__, resp); 1557 #endif 1558 isc_refcount_init(&resp->references, 1); /* DISPENTRY000 */ 1559 1560 if (disp->socktype == isc_socktype_udp) { 1561 isc_result_t result = setup_socket(disp, resp, dest, 1562 &localport); 1563 if (result != ISC_R_SUCCESS) { 1564 isc_mem_put(disp->mctx, resp, sizeof(*resp)); 1565 inc_stats(disp->mgr, dns_resstatscounter_dispsockfail); 1566 return result; 1567 } 1568 } 1569 1570 isc_result_t result = ISC_R_NOMORE; 1571 rcu_read_lock(); 1572 1573 if (disp->socktype == isc_socktype_tcp) { 1574 /* 1575 * TCP dispentries don't use the global QID hash table. 1576 * Responses are matched by scanning disp->active, and 1577 * sequential per-dispatch IDs (bounded by the pipelining 1578 * limit) are guaranteed to be unique within the dispatch. 1579 * FIXEDID TCP dispatches are always fresh and isolated 1580 * (see dns_dispatch_createtcp), so the caller-supplied ID 1581 * can't collide either. 1582 */ 1583 resp->id = ((options & DNS_DISPATCHOPT_FIXEDID) != 0) 1584 ? *idp 1585 : disp->nextid++; 1586 result = ISC_R_SUCCESS; 1587 } else { 1588 size_t i = 0; 1589 do { 1590 /* 1591 * Try somewhat hard to find a unique random ID 1592 * (or use the fixed ID if DNS_DISPATCHOPT_FIXEDID 1593 * is set). 1594 */ 1595 resp->id = ((options & DNS_DISPATCHOPT_FIXEDID) != 0) 1596 ? *idp 1597 : (dns_messageid_t)isc_random16(); 1598 1599 struct cds_lfht_node *node = cds_lfht_add_unique( 1600 disp->mgr->qids, qid_hash(resp), qid_match, 1601 resp, &resp->ht_node); 1602 1603 if (node != &resp->ht_node) { 1604 if ((options & DNS_DISPATCHOPT_FIXEDID) != 0) { 1605 /* 1606 * When using fixed ID, we either 1607 * must use it or fail. 1608 */ 1609 goto fail; 1610 } 1611 } else { 1612 result = ISC_R_SUCCESS; 1613 break; 1614 } 1615 } while (i++ < QID_MAX_TRIES); 1616 } 1617 fail: 1618 if (result != ISC_R_SUCCESS) { 1619 isc_mem_put(disp->mctx, resp, sizeof(*resp)); 1620 rcu_read_unlock(); 1621 return result; 1622 } 1623 1624 isc_mem_attach(disp->mctx, &resp->mctx); 1625 1626 if (transport != NULL) { 1627 dns_transport_attach(transport, &resp->transport); 1628 } 1629 1630 if (tlsctx_cache != NULL) { 1631 isc_tlsctx_cache_attach(tlsctx_cache, &resp->tlsctx_cache); 1632 } 1633 1634 dns_dispatch_attach(disp, &resp->disp); /* DISPATCH001 */ 1635 1636 disp->requests++; 1637 1638 /* 1639 * If this shared TCP dispatch has reached the pipelining limit, 1640 * remove it from the hash table so new queries get a fresh 1641 * connection. The dispatch continues to serve its existing 1642 * queries until they complete. 1643 */ 1644 if (disp->socktype == isc_socktype_tcp && 1645 (disp->options & DNS_DISPATCHOPT_FIXEDID) == 0 && 1646 disp->requests >= dns_dispatch_tcppipelining) 1647 { 1648 (void)cds_lfht_del(disp->mgr->tcps[isc_tid()], &disp->ht_node); 1649 } 1650 1651 inc_stats(disp->mgr, (disp->socktype == isc_socktype_udp) 1652 ? dns_resstatscounter_disprequdp 1653 : dns_resstatscounter_dispreqtcp); 1654 1655 rcu_read_unlock(); 1656 1657 *idp = resp->id; 1658 *respp = resp; 1659 1660 return ISC_R_SUCCESS; 1661 } 1662 1663 isc_result_t 1664 dns_dispatch_getnext(dns_dispentry_t *resp) { 1665 REQUIRE(VALID_RESPONSE(resp)); 1666 REQUIRE(VALID_DISPATCH(resp->disp)); 1667 1668 dns_dispatch_t *disp = resp->disp; 1669 isc_result_t result = ISC_R_SUCCESS; 1670 int32_t timeout = 0; 1671 1672 dispentry_log(resp, ISC_LOG_DEBUG(90), "getnext for QID %d", resp->id); 1673 1674 if (resp->timeout > 0) { 1675 isc_time_t now = isc_loop_now(resp->loop); 1676 timeout = resp->timeout - dispentry_runtime(resp, &now); 1677 if (timeout <= 0) { 1678 return ISC_R_TIMEDOUT; 1679 } 1680 } 1681 1682 REQUIRE(disp->tid == isc_tid()); 1683 switch (disp->socktype) { 1684 case isc_socktype_udp: 1685 udp_dispatch_getnext(resp, timeout); 1686 break; 1687 case isc_socktype_tcp: 1688 tcp_dispatch_getnext(disp, resp, timeout); 1689 break; 1690 default: 1691 UNREACHABLE(); 1692 } 1693 1694 return result; 1695 } 1696 1697 /* 1698 * NOTE: Must be RCU read locked! 1699 */ 1700 static void 1701 udp_dispentry_cancel(dns_dispentry_t *resp, isc_result_t result) { 1702 REQUIRE(VALID_RESPONSE(resp)); 1703 REQUIRE(VALID_DISPATCH(resp->disp)); 1704 REQUIRE(VALID_DISPATCHMGR(resp->disp->mgr)); 1705 1706 dns_dispatch_t *disp = resp->disp; 1707 bool respond = false; 1708 1709 REQUIRE(disp->tid == isc_tid()); 1710 dispentry_log(resp, ISC_LOG_DEBUG(90), 1711 "canceling response: %s, %s/%s (%s/%s), " 1712 "requests %" PRIuFAST32, 1713 isc_result_totext(result), state2str(resp->state), 1714 resp->reading ? "reading" : "not reading", 1715 state2str(disp->state), 1716 disp->reading ? "reading" : "not reading", 1717 disp->requests); 1718 1719 if (ISC_LINK_LINKED(resp, alink)) { 1720 ISC_LIST_UNLINK(disp->active, resp, alink); 1721 } 1722 1723 switch (resp->state) { 1724 case DNS_DISPATCHSTATE_NONE: 1725 break; 1726 1727 case DNS_DISPATCHSTATE_CONNECTING: 1728 break; 1729 1730 case DNS_DISPATCHSTATE_CONNECTED: 1731 if (resp->reading) { 1732 respond = true; 1733 dispentry_log(resp, ISC_LOG_DEBUG(90), 1734 "canceling read on %p", resp->handle); 1735 isc_nm_cancelread(resp->handle); 1736 } 1737 break; 1738 1739 case DNS_DISPATCHSTATE_CANCELED: 1740 goto unlock; 1741 1742 default: 1743 UNREACHABLE(); 1744 } 1745 1746 dec_stats(disp->mgr, dns_resstatscounter_disprequdp); 1747 1748 (void)cds_lfht_del(disp->mgr->qids, &resp->ht_node); 1749 1750 resp->state = DNS_DISPATCHSTATE_CANCELED; 1751 1752 unlock: 1753 if (respond) { 1754 dispentry_log(resp, ISC_LOG_DEBUG(90), "read callback: %s", 1755 isc_result_totext(result)); 1756 resp->response(result, NULL, resp->arg); 1757 } 1758 } 1759 1760 /* 1761 * NOTE: Must be RCU read locked! 1762 */ 1763 static void 1764 tcp_dispentry_cancel(dns_dispentry_t *resp, isc_result_t result) { 1765 REQUIRE(VALID_RESPONSE(resp)); 1766 REQUIRE(VALID_DISPATCH(resp->disp)); 1767 REQUIRE(VALID_DISPATCHMGR(resp->disp->mgr)); 1768 1769 dns_dispatch_t *disp = resp->disp; 1770 dns_displist_t resps = ISC_LIST_INITIALIZER; 1771 1772 REQUIRE(disp->tid == isc_tid()); 1773 dispentry_log(resp, ISC_LOG_DEBUG(90), 1774 "canceling response: %s, %s/%s (%s/%s), " 1775 "requests %" PRIuFAST32, 1776 isc_result_totext(result), state2str(resp->state), 1777 resp->reading ? "reading" : "not reading", 1778 state2str(disp->state), 1779 disp->reading ? "reading" : "not reading", 1780 disp->requests); 1781 1782 switch (resp->state) { 1783 case DNS_DISPATCHSTATE_NONE: 1784 break; 1785 1786 case DNS_DISPATCHSTATE_CONNECTING: 1787 break; 1788 1789 case DNS_DISPATCHSTATE_CONNECTED: 1790 if (resp->reading) { 1791 tcp_recv_add(&resps, resp, ISC_R_CANCELED); 1792 } 1793 1794 INSIST(!ISC_LINK_LINKED(resp, alink)); 1795 1796 if (ISC_LIST_EMPTY(disp->active) && 1797 disp->state == DNS_DISPATCHSTATE_CONNECTED) 1798 { 1799 INSIST(disp->handle != NULL); 1800 1801 /* 1802 * This was the last outstanding response on a still 1803 * connected dispatch. Keep the connection open with a 1804 * read pending (bounded by the idle timeout) so that it 1805 * can be reused by a subsequent query and so that a 1806 * peer-initiated close is noticed promptly instead of 1807 * leaving the socket stuck in CLOSE-WAIT. If the 1808 * dispatch is already shutting down, leave it alone so 1809 * it can be torn down. 1810 */ 1811 if (disp->mgr->reuse_timeout > 0) { 1812 tcp_startrecv_idle(disp); 1813 } else if (disp->reading) { 1814 dispentry_log(resp, ISC_LOG_DEBUG(90), 1815 "canceling read on %p", 1816 disp->handle); 1817 isc_nm_cancelread(disp->handle); 1818 } 1819 } 1820 break; 1821 1822 case DNS_DISPATCHSTATE_CANCELED: 1823 goto unlock; 1824 1825 default: 1826 UNREACHABLE(); 1827 } 1828 1829 dec_stats(disp->mgr, dns_resstatscounter_dispreqtcp); 1830 1831 resp->state = DNS_DISPATCHSTATE_CANCELED; 1832 1833 unlock: 1834 1835 /* 1836 * NOTE: Calling the response callback directly from here should be done 1837 * asynchronously, as the dns_dispatch_done() is usually called directly 1838 * from the response callback, so there's a slight chance that the call 1839 * stack will get higher here, but it's mitigated by the ".reading" 1840 * flag, so we don't ever go into a loop. 1841 */ 1842 1843 tcp_recv_processall(&resps, NULL); 1844 } 1845 1846 static void 1847 dispentry_cancel(dns_dispentry_t *resp, isc_result_t result) { 1848 REQUIRE(VALID_RESPONSE(resp)); 1849 REQUIRE(VALID_DISPATCH(resp->disp)); 1850 1851 dns_dispatch_t *disp = resp->disp; 1852 1853 rcu_read_lock(); 1854 switch (disp->socktype) { 1855 case isc_socktype_udp: 1856 udp_dispentry_cancel(resp, result); 1857 break; 1858 case isc_socktype_tcp: 1859 tcp_dispentry_cancel(resp, result); 1860 break; 1861 default: 1862 UNREACHABLE(); 1863 } 1864 rcu_read_unlock(); 1865 } 1866 1867 void 1868 dns_dispatch_done(dns_dispentry_t **respp) { 1869 REQUIRE(VALID_RESPONSE(*respp)); 1870 1871 dns_dispentry_t *resp = *respp; 1872 *respp = NULL; 1873 1874 dispentry_cancel(resp, ISC_R_CANCELED); 1875 dns_dispentry_detach(&resp); /* DISPENTRY000 */ 1876 } 1877 1878 static void 1879 udp_startrecv(isc_nmhandle_t *handle, dns_dispentry_t *resp) { 1880 REQUIRE(VALID_RESPONSE(resp)); 1881 1882 dispentry_log(resp, ISC_LOG_DEBUG(90), "attaching handle %p to %p", 1883 handle, &resp->handle); 1884 isc_nmhandle_attach(handle, &resp->handle); 1885 dns_dispentry_ref(resp); /* DISPENTRY003 */ 1886 dispentry_log(resp, ISC_LOG_DEBUG(90), "reading"); 1887 isc_nm_read(resp->handle, udp_recv, resp); 1888 resp->reading = true; 1889 } 1890 1891 static void 1892 tcp_startrecv(dns_dispatch_t *disp, dns_dispentry_t *resp) { 1893 REQUIRE(VALID_DISPATCH(disp)); 1894 REQUIRE(disp->socktype == isc_socktype_tcp); 1895 1896 if (disp->reading) { 1897 return; 1898 } 1899 1900 dns_dispatch_ref(disp); /* DISPATCH002 */ 1901 if (resp != NULL) { 1902 dispentry_log(resp, ISC_LOG_DEBUG(90), "reading from %p", 1903 disp->handle); 1904 INSIST(!isc_time_isepoch(&resp->start)); 1905 } else { 1906 dispatch_log(disp, ISC_LOG_DEBUG(90), 1907 "TCP reading without response from %p", 1908 disp->handle); 1909 } 1910 isc_nm_read(disp->handle, tcp_recv, disp); 1911 disp->reading = true; 1912 } 1913 1914 static void 1915 resp_connected(void *arg) { 1916 dns_dispentry_t *resp = arg; 1917 dispentry_log(resp, ISC_LOG_DEBUG(90), "connect callback: %s", 1918 isc_result_totext(resp->result)); 1919 1920 resp->connected(resp->result, NULL, resp->arg); 1921 dns_dispentry_detach(&resp); /* DISPENTRY005 */ 1922 } 1923 1924 static void 1925 tcp_connected(isc_nmhandle_t *handle, isc_result_t eresult, void *arg) { 1926 dns_dispatch_t *disp = (dns_dispatch_t *)arg; 1927 dns_dispentry_t *resp = NULL; 1928 dns_dispentry_t *next = NULL; 1929 dns_displist_t resps = ISC_LIST_INITIALIZER; 1930 1931 if (isc_log_wouldlog(dns_lctx, 90)) { 1932 char localbuf[ISC_SOCKADDR_FORMATSIZE]; 1933 char peerbuf[ISC_SOCKADDR_FORMATSIZE]; 1934 if (handle != NULL) { 1935 isc_sockaddr_t local = isc_nmhandle_localaddr(handle); 1936 isc_sockaddr_t peer = isc_nmhandle_peeraddr(handle); 1937 1938 isc_sockaddr_format(&local, localbuf, 1939 ISC_SOCKADDR_FORMATSIZE); 1940 isc_sockaddr_format(&peer, peerbuf, 1941 ISC_SOCKADDR_FORMATSIZE); 1942 } else { 1943 isc_sockaddr_format(&disp->local, localbuf, 1944 ISC_SOCKADDR_FORMATSIZE); 1945 isc_sockaddr_format(&disp->peer, peerbuf, 1946 ISC_SOCKADDR_FORMATSIZE); 1947 } 1948 1949 dispatch_log(disp, ISC_LOG_DEBUG(90), 1950 "connected from %s to %s: %s", localbuf, peerbuf, 1951 isc_result_totext(eresult)); 1952 } 1953 1954 REQUIRE(disp->tid == isc_tid()); 1955 INSIST(disp->state == DNS_DISPATCHSTATE_CONNECTING); 1956 1957 /* 1958 * If there are pending responses, call the connect 1959 * callbacks for all of them. 1960 */ 1961 for (resp = ISC_LIST_HEAD(disp->pending); resp != NULL; resp = next) { 1962 next = ISC_LIST_NEXT(resp, plink); 1963 ISC_LIST_UNLINK(disp->pending, resp, plink); 1964 ISC_LIST_APPEND(resps, resp, rlink); 1965 resp->result = eresult; 1966 1967 if (resp->state == DNS_DISPATCHSTATE_CANCELED) { 1968 resp->result = ISC_R_CANCELED; 1969 } else if (eresult == ISC_R_SUCCESS) { 1970 resp->state = DNS_DISPATCHSTATE_CONNECTED; 1971 ISC_LIST_APPEND(disp->active, resp, alink); 1972 resp->reading = true; 1973 dispentry_log(resp, ISC_LOG_DEBUG(90), "start reading"); 1974 } else { 1975 resp->state = DNS_DISPATCHSTATE_NONE; 1976 } 1977 } 1978 1979 /* Take the oldest active response. */ 1980 dns_dispentry_t *oldest = ISC_LIST_HEAD(disp->active); 1981 if (oldest == NULL) { 1982 /* All responses have been canceled */ 1983 disp->state = DNS_DISPATCHSTATE_CANCELED; 1984 } else if (eresult == ISC_R_SUCCESS) { 1985 disp->state = DNS_DISPATCHSTATE_CONNECTED; 1986 isc_nmhandle_attach(handle, &disp->handle); 1987 isc_nmhandle_cleartimeout(disp->handle); 1988 if (oldest->timeout != 0) { 1989 isc_nmhandle_settimeout(disp->handle, oldest->timeout); 1990 } 1991 tcp_startrecv(disp, resp); 1992 } else { 1993 disp->state = DNS_DISPATCHSTATE_NONE; 1994 } 1995 1996 for (resp = ISC_LIST_HEAD(resps); resp != NULL; resp = next) { 1997 next = ISC_LIST_NEXT(resp, rlink); 1998 ISC_LIST_UNLINK(resps, resp, rlink); 1999 2000 resp_connected(resp); 2001 } 2002 2003 dns_dispatch_detach(&disp); /* DISPATCH003 */ 2004 } 2005 2006 static void 2007 udp_connected(isc_nmhandle_t *handle, isc_result_t eresult, void *arg) { 2008 dns_dispentry_t *resp = (dns_dispentry_t *)arg; 2009 dns_dispatch_t *disp = resp->disp; 2010 2011 dispentry_log(resp, ISC_LOG_DEBUG(90), "connected: %s", 2012 isc_result_totext(eresult)); 2013 2014 REQUIRE(disp->tid == isc_tid()); 2015 switch (resp->state) { 2016 case DNS_DISPATCHSTATE_CANCELED: 2017 eresult = ISC_R_CANCELED; 2018 ISC_LIST_UNLINK(disp->pending, resp, plink); 2019 goto unlock; 2020 case DNS_DISPATCHSTATE_CONNECTING: 2021 ISC_LIST_UNLINK(disp->pending, resp, plink); 2022 break; 2023 default: 2024 UNREACHABLE(); 2025 } 2026 2027 switch (eresult) { 2028 case ISC_R_CANCELED: 2029 break; 2030 case ISC_R_SUCCESS: 2031 resp->state = DNS_DISPATCHSTATE_CONNECTED; 2032 udp_startrecv(handle, resp); 2033 break; 2034 case ISC_R_NOPERM: 2035 case ISC_R_ADDRINUSE: { 2036 in_port_t localport = isc_sockaddr_getport(&disp->local); 2037 isc_result_t result; 2038 2039 /* probably a port collision; try a different one */ 2040 result = setup_socket(disp, resp, &resp->peer, &localport); 2041 if (result == ISC_R_SUCCESS) { 2042 udp_dispatch_connect(disp, resp); 2043 goto detach; 2044 } 2045 resp->state = DNS_DISPATCHSTATE_NONE; 2046 break; 2047 } 2048 default: 2049 resp->state = DNS_DISPATCHSTATE_NONE; 2050 break; 2051 } 2052 unlock: 2053 2054 dispentry_log(resp, ISC_LOG_DEBUG(90), "connect callback: %s", 2055 isc_result_totext(eresult)); 2056 resp->connected(eresult, NULL, resp->arg); 2057 2058 detach: 2059 dns_dispentry_detach(&resp); /* DISPENTRY004 */ 2060 } 2061 2062 static void 2063 udp_dispatch_connect(dns_dispatch_t *disp, dns_dispentry_t *resp) { 2064 REQUIRE(disp->tid == isc_tid()); 2065 resp->state = DNS_DISPATCHSTATE_CONNECTING; 2066 resp->start = isc_loop_now(resp->loop); 2067 dns_dispentry_ref(resp); /* DISPENTRY004 */ 2068 ISC_LIST_APPEND(disp->pending, resp, plink); 2069 2070 isc_nm_udpconnect(disp->mgr->nm, &resp->local, &resp->peer, 2071 udp_connected, resp, resp->timeout); 2072 } 2073 2074 static inline const char * 2075 get_tls_sni_hostname(dns_dispentry_t *resp) { 2076 char *hostname = NULL; 2077 2078 if (resp->transport != NULL) { 2079 hostname = dns_transport_get_remote_hostname(resp->transport); 2080 } 2081 2082 if (hostname == NULL) { 2083 return NULL; 2084 } 2085 2086 if (isc_tls_valid_sni_hostname(hostname)) { 2087 return hostname; 2088 } 2089 2090 return NULL; 2091 } 2092 2093 static isc_result_t 2094 tcp_dispatch_connect(dns_dispatch_t *disp, dns_dispentry_t *resp) { 2095 dns_transport_type_t transport_type = DNS_TRANSPORT_TCP; 2096 isc_tlsctx_t *tlsctx = NULL; 2097 isc_tlsctx_client_session_cache_t *sess_cache = NULL; 2098 2099 if (resp->transport != NULL) { 2100 transport_type = dns_transport_get_type(resp->transport); 2101 } 2102 2103 if (transport_type == DNS_TRANSPORT_TLS) { 2104 isc_result_t result; 2105 2106 result = dns_transport_get_tlsctx( 2107 resp->transport, &resp->peer, resp->tlsctx_cache, 2108 resp->mctx, &tlsctx, &sess_cache); 2109 2110 if (result != ISC_R_SUCCESS) { 2111 return result; 2112 } 2113 INSIST(tlsctx != NULL); 2114 } 2115 2116 /* Check whether the dispatch is already connecting or connected. */ 2117 REQUIRE(disp->tid == isc_tid()); 2118 switch (disp->state) { 2119 case DNS_DISPATCHSTATE_NONE: 2120 /* First connection, continue with connecting */ 2121 disp->state = DNS_DISPATCHSTATE_CONNECTING; 2122 resp->state = DNS_DISPATCHSTATE_CONNECTING; 2123 resp->start = isc_loop_now(resp->loop); 2124 dns_dispentry_ref(resp); /* DISPENTRY005 */ 2125 ISC_LIST_APPEND(disp->pending, resp, plink); 2126 2127 char localbuf[ISC_SOCKADDR_FORMATSIZE]; 2128 char peerbuf[ISC_SOCKADDR_FORMATSIZE]; 2129 2130 isc_sockaddr_format(&disp->local, localbuf, 2131 ISC_SOCKADDR_FORMATSIZE); 2132 isc_sockaddr_format(&disp->peer, peerbuf, 2133 ISC_SOCKADDR_FORMATSIZE); 2134 2135 dns_dispatch_ref(disp); /* DISPATCH003 */ 2136 dispentry_log(resp, ISC_LOG_DEBUG(90), 2137 "connecting from %s to %s, timeout %u", localbuf, 2138 peerbuf, resp->timeout); 2139 2140 const char *hostname = get_tls_sni_hostname(resp); 2141 2142 isc_nm_streamdnsconnect(disp->mgr->nm, &disp->local, 2143 &disp->peer, tcp_connected, disp, 2144 resp->timeout, tlsctx, hostname, 2145 sess_cache, ISC_NM_PROXY_NONE, NULL); 2146 break; 2147 2148 case DNS_DISPATCHSTATE_CONNECTING: 2149 /* Connection pending; add resp to the list */ 2150 resp->state = DNS_DISPATCHSTATE_CONNECTING; 2151 resp->start = isc_loop_now(resp->loop); 2152 dns_dispentry_ref(resp); /* DISPENTRY005 */ 2153 ISC_LIST_APPEND(disp->pending, resp, plink); 2154 break; 2155 2156 case DNS_DISPATCHSTATE_CONNECTED: 2157 resp->state = DNS_DISPATCHSTATE_CONNECTED; 2158 resp->start = isc_loop_now(resp->loop); 2159 2160 /* Add the resp to the reading list */ 2161 ISC_LIST_APPEND(disp->active, resp, alink); 2162 dispentry_log(resp, ISC_LOG_DEBUG(90), 2163 "already connected; attaching"); 2164 resp->reading = true; 2165 2166 /* 2167 * Replace any idle timeout with this query's timeout. The 2168 * connection may have been kept reading while idle (with the 2169 * idle timeout applied), so update the handle timeout even when 2170 * a read is already pending. 2171 */ 2172 isc_nmhandle_cleartimeout(disp->handle); 2173 if (resp->timeout != 0) { 2174 isc_nmhandle_settimeout(disp->handle, resp->timeout); 2175 } 2176 2177 if (!disp->reading) { 2178 /* Restart the reading */ 2179 isc_nmhandle_cleartimeout(disp->handle); 2180 if (resp->timeout != 0) { 2181 isc_nmhandle_settimeout(disp->handle, 2182 resp->timeout); 2183 } 2184 tcp_startrecv(disp, resp); 2185 } 2186 2187 /* Already connected; call the connected cb asynchronously */ 2188 dns_dispentry_ref(resp); /* DISPENTRY005 */ 2189 isc_async_run(resp->loop, resp_connected, resp); 2190 break; 2191 2192 default: 2193 UNREACHABLE(); 2194 } 2195 2196 return ISC_R_SUCCESS; 2197 } 2198 2199 isc_result_t 2200 dns_dispatch_connect(dns_dispentry_t *resp) { 2201 REQUIRE(VALID_RESPONSE(resp)); 2202 REQUIRE(VALID_DISPATCH(resp->disp)); 2203 2204 dns_dispatch_t *disp = resp->disp; 2205 2206 switch (disp->socktype) { 2207 case isc_socktype_tcp: 2208 return tcp_dispatch_connect(disp, resp); 2209 2210 case isc_socktype_udp: 2211 udp_dispatch_connect(disp, resp); 2212 return ISC_R_SUCCESS; 2213 2214 default: 2215 UNREACHABLE(); 2216 } 2217 } 2218 2219 static void 2220 send_done(isc_nmhandle_t *handle, isc_result_t result, void *cbarg) { 2221 dns_dispentry_t *resp = (dns_dispentry_t *)cbarg; 2222 2223 REQUIRE(VALID_RESPONSE(resp)); 2224 2225 dns_dispatch_t *disp = resp->disp; 2226 2227 REQUIRE(VALID_DISPATCH(disp)); 2228 2229 dispentry_log(resp, ISC_LOG_DEBUG(90), "sent: %s", 2230 isc_result_totext(result)); 2231 2232 resp->sent(result, NULL, resp->arg); 2233 2234 if (result != ISC_R_SUCCESS) { 2235 dispentry_cancel(resp, result); 2236 } 2237 2238 dns_dispentry_detach(&resp); /* DISPENTRY007 */ 2239 isc_nmhandle_detach(&handle); 2240 } 2241 2242 static void 2243 tcp_dispatch_getnext(dns_dispatch_t *disp, dns_dispentry_t *resp, 2244 int32_t timeout) { 2245 REQUIRE(timeout <= INT16_MAX); 2246 2247 dispentry_log(resp, ISC_LOG_DEBUG(90), "continue reading"); 2248 2249 if (!resp->reading) { 2250 ISC_LIST_APPEND(disp->active, resp, alink); 2251 resp->reading = true; 2252 } 2253 2254 if (disp->reading) { 2255 return; 2256 } 2257 2258 if (timeout > 0) { 2259 isc_nmhandle_settimeout(disp->handle, timeout); 2260 } 2261 2262 dns_dispatch_ref(disp); /* DISPATCH002 */ 2263 isc_nm_read(disp->handle, tcp_recv, disp); 2264 disp->reading = true; 2265 } 2266 2267 static void 2268 udp_dispatch_getnext(dns_dispentry_t *resp, int32_t timeout) { 2269 REQUIRE(timeout <= INT16_MAX); 2270 2271 if (resp->reading) { 2272 return; 2273 } 2274 2275 if (timeout > 0) { 2276 isc_nmhandle_settimeout(resp->handle, timeout); 2277 } 2278 2279 dispentry_log(resp, ISC_LOG_DEBUG(90), "continue reading"); 2280 2281 dns_dispentry_ref(resp); /* DISPENTRY003 */ 2282 isc_nm_read(resp->handle, udp_recv, resp); 2283 resp->reading = true; 2284 } 2285 2286 void 2287 dns_dispatch_resume(dns_dispentry_t *resp, uint16_t timeout) { 2288 REQUIRE(VALID_RESPONSE(resp)); 2289 REQUIRE(VALID_DISPATCH(resp->disp)); 2290 2291 dns_dispatch_t *disp = resp->disp; 2292 2293 dispentry_log(resp, ISC_LOG_DEBUG(90), "resume"); 2294 2295 REQUIRE(disp->tid == isc_tid()); 2296 switch (disp->socktype) { 2297 case isc_socktype_udp: { 2298 udp_dispatch_getnext(resp, timeout); 2299 break; 2300 } 2301 case isc_socktype_tcp: 2302 INSIST(disp->timedout > 0); 2303 disp->timedout--; 2304 tcp_dispatch_getnext(disp, resp, timeout); 2305 break; 2306 default: 2307 UNREACHABLE(); 2308 } 2309 } 2310 2311 void 2312 dns_dispatch_send(dns_dispentry_t *resp, isc_region_t *r) { 2313 REQUIRE(VALID_RESPONSE(resp)); 2314 REQUIRE(VALID_DISPATCH(resp->disp)); 2315 2316 dns_dispatch_t *disp = resp->disp; 2317 isc_nmhandle_t *sendhandle = NULL; 2318 2319 dispentry_log(resp, ISC_LOG_DEBUG(90), "sending"); 2320 switch (disp->socktype) { 2321 case isc_socktype_udp: 2322 isc_nmhandle_attach(resp->handle, &sendhandle); 2323 break; 2324 case isc_socktype_tcp: 2325 isc_nmhandle_attach(disp->handle, &sendhandle); 2326 break; 2327 default: 2328 UNREACHABLE(); 2329 } 2330 dns_dispentry_ref(resp); /* DISPENTRY007 */ 2331 isc_nm_send(sendhandle, r, send_done, resp); 2332 } 2333 2334 isc_result_t 2335 dns_dispatch_getlocaladdress(dns_dispatch_t *disp, isc_sockaddr_t *addrp) { 2336 REQUIRE(VALID_DISPATCH(disp)); 2337 REQUIRE(addrp != NULL); 2338 2339 if (disp->socktype == isc_socktype_udp) { 2340 *addrp = disp->local; 2341 return ISC_R_SUCCESS; 2342 } 2343 return ISC_R_NOTIMPLEMENTED; 2344 } 2345 2346 isc_result_t 2347 dns_dispentry_getlocaladdress(dns_dispentry_t *resp, isc_sockaddr_t *addrp) { 2348 REQUIRE(VALID_RESPONSE(resp)); 2349 REQUIRE(VALID_DISPATCH(resp->disp)); 2350 REQUIRE(addrp != NULL); 2351 2352 dns_dispatch_t *disp = resp->disp; 2353 2354 switch (disp->socktype) { 2355 case isc_socktype_tcp: 2356 *addrp = disp->local; 2357 return ISC_R_SUCCESS; 2358 case isc_socktype_udp: 2359 *addrp = isc_nmhandle_localaddr(resp->handle); 2360 return ISC_R_SUCCESS; 2361 default: 2362 UNREACHABLE(); 2363 } 2364 } 2365 2366 dns_dispatch_t * 2367 dns_dispatchset_get(dns_dispatchset_t *dset) { 2368 uint32_t tid = isc_tid(); 2369 2370 /* check that dispatch set is configured */ 2371 if (dset == NULL || dset->ndisp == 0) { 2372 return NULL; 2373 } 2374 2375 INSIST(tid < dset->ndisp); 2376 2377 return dset->dispatches[tid]; 2378 } 2379 2380 isc_result_t 2381 dns_dispatchset_create(isc_mem_t *mctx, dns_dispatch_t *source, 2382 dns_dispatchset_t **dsetp, uint32_t ndisp) { 2383 isc_result_t result; 2384 dns_dispatchset_t *dset = NULL; 2385 dns_dispatchmgr_t *mgr = NULL; 2386 size_t i; 2387 2388 REQUIRE(VALID_DISPATCH(source)); 2389 REQUIRE(source->socktype == isc_socktype_udp); 2390 REQUIRE(dsetp != NULL && *dsetp == NULL); 2391 2392 mgr = source->mgr; 2393 2394 dset = isc_mem_get(mctx, sizeof(dns_dispatchset_t)); 2395 *dset = (dns_dispatchset_t){ .ndisp = ndisp }; 2396 2397 isc_mem_attach(mctx, &dset->mctx); 2398 2399 dset->dispatches = isc_mem_cget(dset->mctx, ndisp, 2400 sizeof(dns_dispatch_t *)); 2401 2402 dset->dispatches[0] = NULL; 2403 dns_dispatch_attach(source, &dset->dispatches[0]); /* DISPATCH004 */ 2404 2405 for (i = 1; i < dset->ndisp; i++) { 2406 result = dispatch_createudp(mgr, &source->local, i, 2407 &dset->dispatches[i]); 2408 if (result != ISC_R_SUCCESS) { 2409 goto fail; 2410 } 2411 } 2412 2413 *dsetp = dset; 2414 2415 return ISC_R_SUCCESS; 2416 2417 fail: 2418 for (size_t j = 0; j < i; j++) { 2419 dns_dispatch_detach(&(dset->dispatches[j])); /* DISPATCH004 */ 2420 } 2421 isc_mem_cput(dset->mctx, dset->dispatches, ndisp, 2422 sizeof(dns_dispatch_t *)); 2423 2424 isc_mem_putanddetach(&dset->mctx, dset, sizeof(dns_dispatchset_t)); 2425 return result; 2426 } 2427 2428 void 2429 dns_dispatchset_destroy(dns_dispatchset_t **dsetp) { 2430 REQUIRE(dsetp != NULL && *dsetp != NULL); 2431 2432 dns_dispatchset_t *dset = *dsetp; 2433 *dsetp = NULL; 2434 2435 for (size_t i = 0; i < dset->ndisp; i++) { 2436 dns_dispatch_detach(&(dset->dispatches[i])); /* DISPATCH004 */ 2437 } 2438 isc_mem_cput(dset->mctx, dset->dispatches, dset->ndisp, 2439 sizeof(dns_dispatch_t *)); 2440 isc_mem_putanddetach(&dset->mctx, dset, sizeof(dns_dispatchset_t)); 2441 } 2442 2443 isc_result_t 2444 dns_dispatch_checkperm(dns_dispatch_t *disp) { 2445 REQUIRE(VALID_DISPATCH(disp)); 2446 2447 if (disp->handle == NULL || disp->socktype == isc_socktype_udp) { 2448 return ISC_R_NOPERM; 2449 } 2450 2451 return isc_nm_xfr_checkperm(disp->handle); 2452 } 2453