diff --git a/docs/cn/HttpClient.md b/docs/cn/HttpClient.md index d67b57ed2..6dffc7d87 100644 --- a/docs/cn/HttpClient.md +++ b/docs/cn/HttpClient.md @@ -21,23 +21,32 @@ class HttpClient { // 获取请求头部 const char* getHeader(const char* key); - // 设置http代理 + // 设置http代理(明文HTTP走绝对URI转发) int setHttpProxy(const char* host, int port); - // 设置https代理 + // 设置https代理(HTTPS走HTTP CONNECT隧道,与目标端到端TLS) int setHttpsProxy(const char* host, int port); // 添加不走代理 int addNoProxy(const char* host); + // 设置代理认证(http转发用Basic头,CONNECT隧道用Proxy-Authorization) + int setProxyAuth(const char* username, const char* password); // 同步发送 int send(HttpRequest* req, HttpResponse* resp); // 异步发送 int sendAsync(HttpRequestPtr req, HttpResponseCallback resp_cb = NULL); + // 按请求中的目标及代理配置建立连接;成功返回时CONNECT隧道和TLS握手均已完成 + int connect(HttpRequest& req); + // 直接连接指定主机 + int connect(const char* host, int port = DEFAULT_HTTP_PORT, int https = 0, int timeout = DEFAULT_HTTP_CONNECT_TIMEOUT); + // 关闭连接 (HttpClient对象析构时会自动调用) int close(); }; +当 `HttpRequest` 已通过 `SetProxy` 或 `SetTunnelProxy` 配置代理时,以请求级配置为准;否则使用 `HttpClient` 的全局代理配置。重定向后的连接目标未变化时复用当前连接,目标或代理变化时使用独立连接完成重定向,不会关闭调用方原有的 keep-alive 连接。 + namespace requests { // 同步请求 diff --git a/docs/cn/HttpMessage.md b/docs/cn/HttpMessage.md index e6cd4e091..f58828a6c 100644 --- a/docs/cn/HttpMessage.md +++ b/docs/cn/HttpMessage.md @@ -80,6 +80,10 @@ class HttpRequest : public HttpMessage { // 设置代理 void SetProxy(const char* host, int port); + // 设置HTTP正向代理Basic认证,username为空时删除Proxy-Authorization + void SetProxyAuth(const char* username, const char* password = NULL); + // 设置HTTP CONNECT隧道代理及握手认证 + void SetTunnelProxy(const char* host, int port, const char* username = NULL, const char* password = NULL); // 设置认证 void SetAuth(const std::string& auth); diff --git a/event/hevent.c b/event/hevent.c index cd554f285..2332e8758 100644 --- a/event/hevent.c +++ b/event/hevent.c @@ -499,8 +499,11 @@ const char* hio_get_hostname(hio_t* io) { int hio_set_proxy(hio_t* io, proxy_setting_t* setting) { if (io == NULL || setting == NULL) return -1; - // only SOCKS5 is implemented so far - if (setting->protocol != PROXY_PROTOCOL_SOCKS5) return -1; + // implemented: SOCKS5, HTTP CONNECT + if (setting->protocol != PROXY_PROTOCOL_SOCKS5 && + setting->protocol != PROXY_PROTOCOL_HTTP_CONNECT) { + return -1; + } if (io->proxy == NULL) { HV_ALLOC_SIZEOF(io->proxy); if (io->proxy == NULL) return -1; diff --git a/event/hloop.h b/event/hloop.h index cc6aec42b..ec7541f97 100644 --- a/event/hloop.h +++ b/event/hloop.h @@ -359,17 +359,19 @@ HV_EXPORT const char* hio_get_hostname(hio_t* io); // already the proxy connection). // // The setting is copied. Leave username empty for no auth, or set -// username/password for auth (SOCKS5 => RFC 1929). Only PROXY_PROTOCOL_SOCKS5 -// is implemented so far. +// username/password for auth (SOCKS5 => RFC 1929, HTTP CONNECT => Basic). +// Implemented protocols: PROXY_PROTOCOL_SOCKS5, PROXY_PROTOCOL_HTTP_CONNECT. // NOTE: set before hio_connect(). typedef enum { - PROXY_PROTOCOL_NONE = 0, - PROXY_PROTOCOL_SOCKS5 = 1, + PROXY_PROTOCOL_NONE = 0, + PROXY_PROTOCOL_SOCKS5 = 1, + PROXY_PROTOCOL_HTTP_CONNECT = 2, // HTTP CONNECT tunnel (RFC 7231 4.3.6) } proxy_protocol_e; typedef struct proxy_setting_s { int protocol; // proxy_protocol_e - char proxy_host[256]; // proxy host (SOCKS5: unused, socket is the proxy) + char proxy_host[256]; // proxy host (unused by the io layer: the socket + // is already the proxy connection; kept for ref) int proxy_port; char target_host[256]; // final target the proxy should CONNECT to int target_port; diff --git a/event/nio.c b/event/nio.c index 202e14519..0366bf4ca 100644 --- a/event/nio.c +++ b/event/nio.c @@ -263,9 +263,9 @@ enum socks5_state_e { static void socks5_handshake(hio_t* io); -static void socks5_fail(hio_t* io) { +static void proxy_fail(hio_t* io) { if (io->error == 0) io->error = ERR_CONNECT; - hlogw("connfd=%d socks5 handshake error", io->fd); + hlogw("connfd=%d proxy handshake error", io->fd); hio_close(io); } @@ -277,16 +277,18 @@ static void socks5_expect(hio_t* io, int state, int want) { s5->want = want; } -// Raw handshake send. The SOCKS5 handshake runs immediately after the TCP +// Raw handshake send. The proxy handshake runs immediately after the TCP // connection to the proxy is established, when the socket send buffer is empty -// and the messages are tiny (<= 513 bytes), so a short write is not expected. +// and the message is tiny (SOCKS5 <= 513 bytes; HTTP CONNECT < ~1.3KB), far +// smaller than the default send buffer, so a single send() transfers it all. // We deliberately do NOT use hio_write() here: it would invoke the upper-layer // write_cb (leaking handshake bytes, including credentials, to the application // before onConnection), dispatch to hssl_write() with a not-yet-created SSL // handle for a TLS target, and enqueue on EAGAIN via hio_add() which would -// clobber the handshake read handler. A short write or error is treated as -// fatal and closes the connection. -static int socks5_send(hio_t* io, const void* buf, int len) { +// clobber the handshake read handler (io has a single cb slot). A short write +// cannot happen here in practice; if it somehow does, it is treated as fatal +// (return -1) rather than blocking the event loop. +static int proxy_send(hio_t* io, const void* buf, int len) { int flag = 0; #ifdef MSG_NOSIGNAL flag |= MSG_NOSIGNAL; @@ -300,15 +302,15 @@ static void socks5_send_connect(hio_t* io) { proxy_conn_t* s5 = io->proxy; unsigned char buf[300]; int n = socks5_build_connect_request(s5, buf); - if (n < 0) { socks5_fail(io); return; } - if (socks5_send(io, buf, n) != 0) { socks5_fail(io); return; } + if (n < 0) { proxy_fail(io); return; } + if (proxy_send(io, buf, n) != 0) { proxy_fail(io); return; } socks5_expect(io, S5_RECV_REPLY_HEAD, 4); } // hand off the established proxy tunnel to the upper layer: stop the handshake // read handler, then run the SSL handshake / connect_cb. io->read_cb was never // touched, so the upper-layer Channel read callback stays intact. -static void socks5_established(hio_t* io) { +static void proxy_established(hio_t* io) { hio_del(io, HV_READ); nio_connect_established(io); } @@ -321,29 +323,29 @@ static void socks5_dispatch(hio_t* io) { switch (s5->state) { case S5_RECV_METHOD: // VER METHOD - if (buf[0] != SOCKS5_VERSION) { socks5_fail(io); return; } + if (buf[0] != SOCKS5_VERSION) { proxy_fail(io); return; } if (buf[1] == SOCKS5_AUTH_NONE) { socks5_send_connect(io); } else if (buf[1] == SOCKS5_AUTH_USERPASS && s5->setting.username[0]) { unsigned char req[640]; int n = socks5_build_auth_request(s5, req); - if (socks5_send(io, req, n) != 0) { socks5_fail(io); return; } + if (proxy_send(io, req, n) != 0) { proxy_fail(io); return; } socks5_expect(io, S5_RECV_AUTH, 2); } else { - socks5_fail(io); // no acceptable method + proxy_fail(io); // no acceptable method } return; case S5_RECV_AUTH: // VER STATUS (0 == success) - if (buf[1] != 0x00) { socks5_fail(io); return; } + if (buf[1] != 0x00) { proxy_fail(io); return; } socks5_send_connect(io); return; case S5_RECV_REPLY_HEAD: { // VER REP RSV ATYP - if (buf[0] != SOCKS5_VERSION) { socks5_fail(io); return; } - if (buf[1] != SOCKS5_REP_SUCCESS) { io->error = ERR_CONNECT; socks5_fail(io); return; } + if (buf[0] != SOCKS5_VERSION) { proxy_fail(io); return; } + if (buf[1] != SOCKS5_REP_SUCCESS) { io->error = ERR_CONNECT; proxy_fail(io); return; } unsigned char atyp = buf[3]; if (atyp == SOCKS5_ATYP_IPV4) { socks5_expect(io, S5_RECV_REPLY_ADDR, 4 + 2); // addr + port @@ -354,14 +356,14 @@ static void socks5_dispatch(hio_t* io) { // first requiring the length byte. socks5_expect(io, S5_RECV_REPLY_DADDR, 1); } else { - socks5_fail(io); + proxy_fail(io); } return; } case S5_RECV_REPLY_ADDR: // bound addr+port consumed; tunnel is up - socks5_established(io); + proxy_established(io); return; case S5_RECV_REPLY_DADDR: @@ -372,11 +374,11 @@ static void socks5_dispatch(hio_t* io) { socks5_expect(io, S5_RECV_REPLY_DADDR, dlen + 2); return; } - socks5_established(io); + proxy_established(io); return; default: - socks5_fail(io); + proxy_fail(io); return; } } @@ -387,14 +389,14 @@ static void socks5_handshake(hio_t* io) { proxy_conn_t* s5 = io->proxy; while (s5->rlen < s5->want) { int need = s5->want - s5->rlen; - if (s5->want > (int)sizeof(s5->rbuf)) { socks5_fail(io); return; } + if (s5->want > (int)sizeof(s5->rbuf)) { proxy_fail(io); return; } int n = recv(io->fd, (char*)s5->rbuf + s5->rlen, need, 0); - if (n == 0) { socks5_fail(io); return; } // peer closed + if (n == 0) { proxy_fail(io); return; } // peer closed if (n < 0) { int err = socket_errno(); if (err == EAGAIN || err == EINTR) return; // wait for more io->error = err; - socks5_fail(io); + proxy_fail(io); return; } s5->rlen += n; @@ -407,17 +409,100 @@ static void socks5_handshake_start(hio_t* io) { proxy_conn_t* s5 = io->proxy; unsigned char buf[8]; int n = socks5_build_method_request(s5, buf); - if (socks5_send(io, buf, n) != 0) { socks5_fail(io); return; } + if (proxy_send(io, buf, n) != 0) { proxy_fail(io); return; } socks5_expect(io, S5_RECV_METHOD, 2); hio_add(io, socks5_handshake, HV_READ); } -// Dispatch the proxy handshake by protocol (only SOCKS5 implemented so far). +// HTTP CONNECT handshake (RFC 7231 4.3.6): send a CONNECT request, then read +// response headers until the blank line "\r\n\r\n". A 2xx status establishes +// the tunnel. Like the SOCKS5 handshake this uses a dedicated recv() via +// hio_add (never touches io->read_cb) and is robust to fragmentation. +// +// CONNECT responses carry no body, but a server-first origin protocol (SMTP, +// IMAP, FTP...) may send its greeting immediately after the tunnel opens, so +// those bytes can arrive in the same segment as the response headers. To avoid +// swallowing them, we MSG_PEEK to locate the header terminator, then drain +// EXACTLY the header bytes with a real recv(); anything after "\r\n\r\n" stays +// in the socket for the upper-layer read path. +static void http_connect_handshake(hio_t* io) { + proxy_conn_t* p = io->proxy; + for (;;) { + int cap = (int)sizeof(p->rbuf) - p->rlen; + if (cap <= 0) { proxy_fail(io); return; } // headers too large + // peek (non-destructive): inspect what is available without consuming. + int n = recv(io->fd, (char*)p->rbuf + p->rlen, cap, MSG_PEEK); + if (n == 0) { proxy_fail(io); return; } // peer closed + if (n < 0) { + int err = socket_errno(); + if (err == EAGAIN || err == EINTR) return; // wait for more + io->error = err; + proxy_fail(io); + return; + } + int have = p->rlen + n; + // search for "\r\n\r\n" in the peeked window (rescan from a safe offset) + int start = p->rlen >= 3 ? p->rlen - 3 : 0; + int term = -1; + for (int i = start + 3; i < have; ++i) { + if (p->rbuf[i-3]=='\r' && p->rbuf[i-2]=='\n' && + p->rbuf[i-1]=='\r' && p->rbuf[i]=='\n') { term = i; break; } + } + if (term < 0) { + // no full header yet: consume the peeked bytes into the accumulator + // (they are all header bytes) and keep reading. + int got = recv(io->fd, (char*)p->rbuf + p->rlen, n, 0); + if (got <= 0) { proxy_fail(io); return; } + p->rlen += got; + continue; + } + // full header present. Drain exactly up to and including the terminator, + // leaving any trailing tunnel/greeting bytes in the socket. + int header_len = term + 1; // bytes from socket start + int to_drain = header_len - p->rlen; // not yet consumed + if (to_drain > 0) { + int got = recv(io->fd, (char*)p->rbuf + p->rlen, to_drain, 0); + if (got != to_drain) { proxy_fail(io); return; } + p->rlen += got; + } + // parse status line: "HTTP/1.x SP CODE SP ..." + int code = 0; + char* sp = (char*)memchr(p->rbuf, ' ', p->rlen); + if (sp) code = atoi(sp + 1); + if (code >= 200 && code < 300) { + proxy_established(io); + } else { + hlogw("connfd=%d http proxy CONNECT failed: %d", io->fd, code); + io->error = ERR_CONNECT; + proxy_fail(io); + } + return; + } +} + +// Kick off the HTTP CONNECT handshake once the TCP connection to the proxy is up. +static void http_connect_start(hio_t* io) { + proxy_conn_t* p = io->proxy; + // Max request: "CONNECT " + authority(<=262) + " HTTP/1.1\r\nHost: " + + // authority + "\r\nProxy-Authorization: Basic " + base64(255:255)=~684 + + // "\r\n\r\n" ~= 1.3KB. 2048 leaves headroom. + char buf[2048]; + int n = http_connect_build_request(p, buf, (int)sizeof(buf)); + if (n < 0) { proxy_fail(io); return; } + if (proxy_send(io, buf, n) != 0) { proxy_fail(io); return; } + p->rlen = 0; + hio_add(io, http_connect_handshake, HV_READ); +} + +// Dispatch the proxy handshake by protocol. static void proxy_handshake_start(hio_t* io) { switch (io->proxy->setting.protocol) { case PROXY_PROTOCOL_SOCKS5: socks5_handshake_start(io); return; + case PROXY_PROTOCOL_HTTP_CONNECT: + http_connect_start(io); + return; default: io->error = ERR_INVALID_PARAM; hio_close(io); diff --git a/event/socks5.c b/event/socks5.c index c924e7b69..5ab7a0026 100644 --- a/event/socks5.c +++ b/event/socks5.c @@ -1,9 +1,43 @@ #include "socks5.h" #include +#include #include "hsocket.h" // is_ipv4 / is_ipv6 / inet_pton via hplatform +// Minimal base64 encoder for the HTTP CONNECT Proxy-Authorization header. +// NOTE: implemented locally (not via util/base64.h) because the event layer +// must not depend on util/ (core builds only add -I. -Ibase -Issl -Ievent). +// Writes ceil(len/3)*4 bytes to out (no NUL terminator); returns bytes written. +static int socks5_base64_encode(const unsigned char* in, int len, char* out) { + static const char tbl[] = + "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + int n = 0, i = 0; + while (i + 3 <= len) { + unsigned v = (in[i] << 16) | (in[i+1] << 8) | in[i+2]; + out[n++] = tbl[(v >> 18) & 0x3F]; + out[n++] = tbl[(v >> 12) & 0x3F]; + out[n++] = tbl[(v >> 6) & 0x3F]; + out[n++] = tbl[v & 0x3F]; + i += 3; + } + int rem = len - i; + if (rem == 1) { + unsigned v = in[i] << 16; + out[n++] = tbl[(v >> 18) & 0x3F]; + out[n++] = tbl[(v >> 12) & 0x3F]; + out[n++] = '='; + out[n++] = '='; + } else if (rem == 2) { + unsigned v = (in[i] << 16) | (in[i+1] << 8); + out[n++] = tbl[(v >> 18) & 0x3F]; + out[n++] = tbl[(v >> 12) & 0x3F]; + out[n++] = tbl[(v >> 6) & 0x3F]; + out[n++] = '='; + } + return n; +} + // Build the SOCKS5 method-selection request. // +----+----------+----------+ // |VER | NMETHODS | METHODS | @@ -74,3 +108,44 @@ int socks5_build_connect_request(const proxy_conn_t* s5, unsigned char* buf) { buf[n++] = (unsigned char)(port & 0xFF); return n; } + +// Build an HTTP CONNECT request (RFC 7231 4.3.6). The request-target is the +// authority form "host:port"; an IPv6 literal is bracketed ("[addr]:port") per +// RFC 3986. A Basic Proxy-Authorization header is added when credentials are +// present. +int http_connect_build_request(const proxy_conn_t* p, char* buf, int bufsize) { + const char* host = p->setting.target_host; + int port = p->setting.target_port; + // bracket IPv6 literals in authority form + char authority[300]; + if (is_ipv6(host)) { + snprintf(authority, sizeof(authority), "[%s]:%d", host, port); + } else { + snprintf(authority, sizeof(authority), "%s:%d", host, port); + } + int n = 0; + int r = snprintf(buf + n, bufsize - n, + "CONNECT %s HTTP/1.1\r\nHost: %s\r\n", + authority, authority); + if (r < 0 || r >= bufsize - n) return -1; + n += r; + + if (p->setting.username[0]) { + // credentials = "user:pass" + char cred[520]; + int c = snprintf(cred, sizeof(cred), "%s:%s", + p->setting.username, p->setting.password); + if (c < 0 || c >= (int)sizeof(cred)) return -1; + char b64[768]; + int b = socks5_base64_encode((const unsigned char*)cred, c, b64); + b64[b] = '\0'; + r = snprintf(buf + n, bufsize - n, "Proxy-Authorization: Basic %s\r\n", b64); + if (r < 0 || r >= bufsize - n) return -1; + n += r; + } + + r = snprintf(buf + n, bufsize - n, "\r\n"); + if (r < 0 || r >= bufsize - n) return -1; + n += r; + return n; +} diff --git a/event/socks5.h b/event/socks5.h index 9ca2a4c80..1e4d0af4d 100644 --- a/event/socks5.h +++ b/event/socks5.h @@ -29,15 +29,17 @@ #define SOCKS5_REP_SUCCESS 0x00 // Internal per-connection runtime state for the proxy handshake (held on -// hio_t). Not part of the public configuration. +// hio_t). Not part of the public configuration. Shared by SOCKS5 and HTTP +// CONNECT. typedef struct proxy_conn_s { proxy_setting_t setting; // copied proxy config (target + auth) int state; // socks5_state_e (see nio.c) - // handshake read accumulator: SOCKS5 replies may be fragmented across TCP + // handshake read accumulator: replies may be fragmented across TCP // segments, so bytes are buffered here until a full message is available. - unsigned char rbuf[300]; // max reply: 4 + 1 + 255 + 2 (domain bind) + // SOCKS5 max reply is small; HTTP CONNECT response headers can be larger. + unsigned char rbuf[1024]; int rlen; // bytes currently in rbuf - int want; // bytes needed to complete the current step + int want; // bytes needed to complete the current step (SOCKS5) } proxy_conn_t; BEGIN_EXTERN_C @@ -47,6 +49,14 @@ int socks5_build_method_request (const proxy_conn_t* s5, unsigned char* buf); int socks5_build_auth_request (const proxy_conn_t* s5, unsigned char* buf); int socks5_build_connect_request(const proxy_conn_t* s5, unsigned char* buf); +// Build an HTTP CONNECT request into buf (size bufsize). Sends +// CONNECT target_host:target_port HTTP/1.1 +// Host: target_host:target_port +// [Proxy-Authorization: Basic base64(user:pass)] +// (blank line) +// Returns bytes written (<0 on error / truncation). +int http_connect_build_request(const proxy_conn_t* p, char* buf, int bufsize); + END_EXTERN_C #endif // HV_SOCKS5_H_ diff --git a/http/HttpMessage.cpp b/http/HttpMessage.cpp index fa86c7800..c2450908d 100644 --- a/http/HttpMessage.cpp +++ b/http/HttpMessage.cpp @@ -660,6 +660,10 @@ void HttpRequest::Init() { redirect = 1; proxy = 0; cancel = 0; + tunnel_proxy_host.clear(); + tunnel_proxy_port = 0; + tunnel_proxy_username.clear(); + tunnel_proxy_password.clear(); } void HttpRequest::Reset() { @@ -775,6 +779,26 @@ void HttpRequest::SetProxy(const char* host, int port) { this->host = host; this->port = port; proxy = 1; + // mutually exclusive with the CONNECT-tunnel mode + tunnel_proxy_host.clear(); + tunnel_proxy_port = 0; + tunnel_proxy_username.clear(); + tunnel_proxy_password.clear(); +} + +void HttpRequest::SetProxyAuth(const char* username, const char* password) { + if (username == NULL || *username == '\0') { + headers.erase("Proxy-Authorization"); + return; + } + + std::string credentials = username; + credentials += ':'; + if (password) { + credentials += password; + } + headers["Proxy-Authorization"] = + "Basic " + hv::Base64Encode((const unsigned char*)credentials.data(), credentials.size()); } void HttpRequest::SetAuth(const std::string& auth) { diff --git a/http/HttpMessage.h b/http/HttpMessage.h index 5be241957..8c583f4f8 100644 --- a/http/HttpMessage.h +++ b/http/HttpMessage.h @@ -386,8 +386,16 @@ class HV_EXPORT HttpRequest : public HttpMessage { uint32_t retry_count; uint32_t retry_delay; // unit: ms unsigned redirect: 1; - unsigned proxy : 1; + unsigned proxy : 1; // absolute-URI forward proxy (plain HTTP) unsigned cancel : 1; + // CONNECT-tunnel proxy (for https-over-proxy): when tunnel_proxy_host is set, + // the client connects to the proxy and issues an HTTP CONNECT to + // host:port, then does TLS end-to-end against the origin. Distinct from the + // `proxy` bit above, which is the plain-HTTP absolute-URI forward proxy. + std::string tunnel_proxy_host; + int tunnel_proxy_port; + std::string tunnel_proxy_username; + std::string tunnel_proxy_password; HttpRequest(); @@ -447,6 +455,24 @@ class HV_EXPORT HttpRequest : public HttpMessage { void SetProxy(const char* host, int port); bool IsProxy() { return proxy; } + // Basic authentication for an HTTP forward proxy request. + // Empty username removes Proxy-Authorization. + void SetProxyAuth(const char* username, const char* password = NULL); + + // CONNECT-tunnel proxy (used for https-over-proxy). Unlike SetProxy (plain + // HTTP absolute-URI forwarding), this connects to the proxy and issues an + // HTTP CONNECT to the origin, then does end-to-end TLS with the origin. + // Mutually exclusive with the forward-proxy mode: clears the `proxy` bit. + void SetTunnelProxy(const char* host, int port, + const char* username = NULL, const char* password = NULL) { + proxy = 0; // not an absolute-URI forward proxy + headers.erase("Proxy-Authorization"); + tunnel_proxy_host = host ? host : ""; + tunnel_proxy_port = port; + tunnel_proxy_username = username ? username : ""; + tunnel_proxy_password = password ? password : ""; + } + bool IsTunnelProxy() { return !tunnel_proxy_host.empty(); } // Auth void SetAuth(const std::string& auth); diff --git a/http/client/AsyncHttpClient.cpp b/http/client/AsyncHttpClient.cpp index e16073aa9..991be091e 100644 --- a/http/client/AsyncHttpClient.cpp +++ b/http/client/AsyncHttpClient.cpp @@ -15,45 +15,79 @@ int AsyncHttpClient::send(const HttpRequestPtr& req, HttpResponseCallback resp_c return send(task); } -// createsocket => startConnect => -// onconnect => sendRequest => startRead => -// onread => HttpParser => resp_cb -int AsyncHttpClient::doTask(const HttpClientTaskPtr& task) { +int AsyncHttpClient::checkTaskCancelOrTimeout(const HttpClientTaskPtr& task) { const HttpRequestPtr& req = task->req; if (req->cancel) { return -1; } - // queueInLoop timeout? uint64_t now_hrtime = hloop_now_hrtime(EventLoopThread::hloop()); - int elapsed_ms = (now_hrtime - task->start_time) / 1000; + int elapsed = (now_hrtime - task->start_time) / 1000; int timeout_ms = req->timeout * 1000; - if (timeout_ms > 0 && elapsed_ms >= timeout_ms) { + if (timeout_ms > 0 && elapsed >= timeout_ms) { hlogw("%s queueInLoop timeout!", req->url.c_str()); return -10; } + return elapsed; +} + +// ParseUrl -> if conn_key in conn_pools: getChannel -> startTask -> sendRequest +// -> else: resolveDns -> addChannel -> startTask -> startConnect -> onconnect -> sendRequest +int AsyncHttpClient::doTask(const HttpClientTaskPtr& task) { + int elapsed_ms = checkTaskCancelOrTimeout(task); + if (elapsed_ms < 0) { + return elapsed_ms; + } + const HttpRequestPtr& req = task->req; req->ParseUrl(); + + // The pool is keyed by the logical transport identity (host/port/proxy/TLS), + // not the resolved sockaddr. Reuse an idle connection before DNS so a pool + // hit never waits for or starts an unnecessary resolver query. + int connfd = -1; + auto iter = conn_pools.find(HttpConnKey(*req)); + if (iter != conn_pools.end()) { + iter->second.get(connfd); + } + if (connfd >= 0) { + const SocketChannelPtr& channel = getChannel(connfd); + if (channel) { + int err = startTask(task, channel); + if (err != 0) { + return err; + } + return sendRequest(channel); + } + } + + // Where to open the TCP connection: normally the origin, but for an HTTP + // CONNECT tunnel (https over proxy) it is the proxy. The origin is then + // reached via the proxy's CONNECT (see startConnect / hio_set_proxy). const char* host = req->host.c_str(); + int port = req->port; + if (req->IsTunnelProxy()) { + host = req->tunnel_proxy_host.c_str(); + port = req->tunnel_proxy_port; + } // If host is a numeric IP (or UDS), resolve synchronously (fast path). // Otherwise resolve the hostname asynchronously via EventLoop::resolveDns // so the event loop is never blocked by getaddrinfo. resolveDns returns a // use-after-free-proof DnsID and owns the underlying hdns_t lifetime. - if (req->port < 0 || is_ipaddr(host)) { + if (port < 0 || is_ipaddr(host)) { sockaddr_u peeraddr; memset(&peeraddr, 0, sizeof(peeraddr)); - int ret = sockaddr_set_ipport(&peeraddr, host, req->port); + int ret = sockaddr_set_ipport(&peeraddr, host, port); if (ret != 0) { hloge("unknown host %s", host); return -20; } - return doTaskWithAddr(task, &peeraddr); + return startConnect(task, &peeraddr); } hdns_setting_t opt; if (req->connect_timeout > 0) opt.timeout_ms = req->connect_timeout * 1000; - int port = req->port; DnsID id = EventLoopThread::loop()->resolveDns(host, [this, task, port](int status, int naddrs, const sockaddr_u* addrs) { if (status != HDNS_STATUS_OK || naddrs <= 0) { @@ -63,7 +97,7 @@ int AsyncHttpClient::doTask(const HttpClientTaskPtr& task) { } sockaddr_u peeraddr = addrs[0]; sockaddr_set_port(&peeraddr, port); - int err = doTaskWithAddr(task, &peeraddr); + int err = startConnect(task, &peeraddr); if (err != 0 && task->cb) { task->cb(NULL); } @@ -75,58 +109,48 @@ int AsyncHttpClient::doTask(const HttpClientTaskPtr& task) { return 0; } -// Continue the request once the peer address is known. -int AsyncHttpClient::doTaskWithAddr(const HttpClientTaskPtr& task, const sockaddr_u* paddr) { - const HttpRequestPtr& req = task->req; - if (req->cancel) { - return -1; +// socket -> addChannel -> configure callbacks -> startTask -> startConnect +int AsyncHttpClient::startConnect(const HttpClientTaskPtr& task, const sockaddr_u* paddr) { + int elapsed_ms = checkTaskCancelOrTimeout(task); + if (elapsed_ms < 0) { + return elapsed_ms; } - uint64_t now_hrtime = hloop_now_hrtime(EventLoopThread::hloop()); - int elapsed_ms = (now_hrtime - task->start_time) / 1000; - int timeout_ms = req->timeout * 1000; - if (timeout_ms > 0 && elapsed_ms >= timeout_ms) { - hlogw("%s queueInLoop timeout!", req->url.c_str()); - return -10; - } - - const char* host = req->host.c_str(); + const HttpRequestPtr& req = task->req; sockaddr_u peeraddr = *paddr; - int connfd = -1; - // first get from conn_pools - char strAddr[SOCKADDR_STRLEN] = {0}; - SOCKADDR_STR(&peeraddr, strAddr); - auto iter = conn_pools.find(strAddr); - if (iter != conn_pools.end()) { - // hlogd("get from conn_pools"); - iter->second.get(connfd); - } - + // create socket + int connfd = socket(peeraddr.sa.sa_family, SOCK_STREAM, 0); if (connfd < 0) { - // create socket - connfd = socket(peeraddr.sa.sa_family, SOCK_STREAM, 0); - if (connfd < 0) { - perror("socket"); - return -30; + perror("socket"); + return -30; + } + hio_t* connio = hio_get(EventLoopThread::hloop(), connfd); + assert(connio != NULL); + hio_set_peeraddr(connio, &peeraddr.sa, sockaddr_len(&peeraddr)); + const SocketChannelPtr& channel = addChannel(connio); + channel->getContext()->conn_key = HttpConnKey(*req); + // https over proxy: HTTP CONNECT tunnel to the origin, then TLS with it. + if (req->IsTunnelProxy()) { + proxy_setting_t proxy; + proxy.protocol = PROXY_PROTOCOL_HTTP_CONNECT; + hv_strncpy(proxy.target_host, req->host.c_str(), sizeof(proxy.target_host)); + proxy.target_port = req->port; + if (!req->tunnel_proxy_username.empty()) { + hv_strncpy(proxy.username, req->tunnel_proxy_username.c_str(), sizeof(proxy.username)); + hv_strncpy(proxy.password, req->tunnel_proxy_password.c_str(), sizeof(proxy.password)); } - hio_t* connio = hio_get(EventLoopThread::hloop(), connfd); - assert(connio != NULL); - hio_set_peeraddr(connio, &peeraddr.sa, sockaddr_len(&peeraddr)); - addChannel(connio); - // https - if (req->IsHttps() && !req->IsProxy()) { - hio_enable_ssl(connio); - if (!is_ipaddr(host)) { - hio_set_hostname(connio, host); - } + hio_set_proxy(connio, &proxy); + } + // https: enable TLS against the origin (also for the tunnel case, run + // after the CONNECT handshake completes, with SNI = origin host). + if (req->IsHttps()) { + hio_enable_ssl(connio); + if (!is_ipaddr(req->host.c_str())) { + hio_set_hostname(connio, req->host.c_str()); } } - const SocketChannelPtr& channel = getChannel(connfd); - assert(channel != NULL); - HttpClientContext* ctx = channel->getContext(); - ctx->task = task; channel->onconnect = [&channel]() { sendRequest(channel); }; @@ -177,7 +201,7 @@ int AsyncHttpClient::doTaskWithAddr(const HttpClientTaskPtr& task, const sockadd if (keepalive) { // NOTE: add into conn_pools to reuse // hlogd("add into conn_pools"); - conn_pools[channel->peeraddr()].add(channel->fd()); + conn_pools[ctx->conn_key].add(channel->fd()); } else { channel->close(); } @@ -187,9 +211,12 @@ int AsyncHttpClient::doTaskWithAddr(const HttpClientTaskPtr& task, const sockadd HttpClientContext* ctx = channel->getContext(); // NOTE: remove from conn_pools // hlogd("remove from conn_pools"); - auto iter = conn_pools.find(channel->peeraddr()); + auto iter = conn_pools.find(ctx->conn_key); if (iter != conn_pools.end()) { iter->second.remove(channel->fd()); + if (iter->second.size() == 0) { + conn_pools.erase(iter); + } } const HttpClientTaskPtr& task = ctx->task; @@ -219,6 +246,31 @@ int AsyncHttpClient::doTaskWithAddr(const HttpClientTaskPtr& task, const sockadd removeChannel(channel); }; + int err = startTask(task, channel); + if (err != 0) { + return err; + } + + if (req->connect_timeout > 0) { + channel->setConnectTimeout(req->connect_timeout * 1000); + } + return channel->startConnect(); +} + +int AsyncHttpClient::startTask(const HttpClientTaskPtr& task, + const SocketChannelPtr& channel) { + int elapsed_ms = checkTaskCancelOrTimeout(task); + if (elapsed_ms < 0) { + return elapsed_ms; + } + + const HttpRequestPtr& req = task->req; + int timeout_ms = req->timeout * 1000; + + assert(channel != NULL); + HttpClientContext* ctx = channel->getContext(); + ctx->task = task; + // timer if (timeout_ms > 0) { ctx->timerID = setTimeout(timeout_ms - elapsed_ms, [&channel](TimerID timerID){ @@ -232,17 +284,6 @@ int AsyncHttpClient::doTaskWithAddr(const HttpClientTaskPtr& task, const sockadd }); } - if (channel->isConnected()) { - // sendRequest - sendRequest(channel); - } else { - // startConnect - if (req->connect_timeout > 0) { - channel->setConnectTimeout(req->connect_timeout * 1000); - } - channel->startConnect(); - } - return 0; } diff --git a/http/client/AsyncHttpClient.h b/http/client/AsyncHttpClient.h index 64eb0dc9f..e186925d8 100644 --- a/http/client/AsyncHttpClient.h +++ b/http/client/AsyncHttpClient.h @@ -3,6 +3,7 @@ #include #include +#include #include "EventLoopThread.h" #include "Channel.h" @@ -55,8 +56,67 @@ struct HttpClientTask { }; typedef std::shared_ptr HttpClientTaskPtr; +struct HttpConnKey { + std::string target_host; + int target_port; + std::string proxy_host; + int proxy_port; + bool tls; + + HttpConnKey() + : target_port(0) + , proxy_port(0) + , tls(false) + {} + + explicit HttpConnKey(const HttpRequest& req) + : HttpConnKey() + { + tls = !req.proxy && (req.scheme.compare(0, 5, "https") == 0 || + req.url.compare(0, 8, "https://") == 0); + + if (req.proxy) { + proxy_host = req.host; + proxy_port = req.port; + } else { + target_host = req.host; + target_port = req.port; + if (!req.tunnel_proxy_host.empty()) { + proxy_host = req.tunnel_proxy_host; + proxy_port = req.tunnel_proxy_port; + } + } + } + + static HttpConnKey Direct(const char* host, int port, bool tls) { + HttpConnKey key; + key.target_host = host ? host : ""; + key.target_port = port; + key.tls = tls; + return key; + } + + bool operator==(const HttpConnKey& rhs) const { + return target_port == rhs.target_port && + proxy_port == rhs.proxy_port && + tls == rhs.tls && + target_host == rhs.target_host && + proxy_host == rhs.proxy_host; + } + + bool operator!=(const HttpConnKey& rhs) const { + return !(*this == rhs); + } + + bool operator<(const HttpConnKey& rhs) const { + return std::tie(target_host, target_port, proxy_host, proxy_port, tls) < + std::tie(rhs.target_host, rhs.target_port, rhs.proxy_host, rhs.proxy_port, rhs.tls); + } +}; + struct HttpClientContext { HttpClientTaskPtr task; + HttpConnKey conn_key; HttpResponsePtr resp; HttpParserPtr parser; @@ -143,8 +203,15 @@ class HV_EXPORT AsyncHttpClient : private EventLoopThread { } int doTask(const HttpClientTaskPtr& task); - // @internal: continue doTask after the peer address is known. - int doTaskWithAddr(const HttpClientTaskPtr& task, const sockaddr_u* peeraddr); + // Create/configure a new channel, bind its first task and start connecting. + int startConnect(const HttpClientTaskPtr& task, const sockaddr_u* peeraddr); + + // Bind the current task and arm its remaining end-to-end timeout. + int startTask(const HttpClientTaskPtr& task, const SocketChannelPtr& channel); + + // Return elapsed milliseconds when task is still runnable, -1 when + // cancelled and -10 when its end-to-end timeout has expired. + int checkTaskCancelOrTimeout(const HttpClientTaskPtr& task); static int sendRequest(const SocketChannelPtr& channel); @@ -172,8 +239,8 @@ class HV_EXPORT AsyncHttpClient : private EventLoopThread { // NOTE: just one loop thread, no need mutex. // fd => SocketChannelPtr std::map channels; - // peeraddr => ConnPool - std::map> conn_pools; + // transport identity => ConnPool + std::map> conn_pools; }; } diff --git a/http/client/HttpClient.cpp b/http/client/HttpClient.cpp index 9917dd2d5..1a4b8b591 100644 --- a/http/client/HttpClient.cpp +++ b/http/client/HttpClient.cpp @@ -10,8 +10,10 @@ #include "hlog.h" #include "htime.h" #include "hstring.h" +#include "hurl.h" #include "hsocket.h" #include "hssl.h" +#include "base64.h" #include "HttpParser.h" // for async @@ -32,6 +34,9 @@ struct http_client_s { // https_proxy std::string https_proxy_host; int https_proxy_port; + // proxy auth (Basic for http, SOCKS5/CONNECT tunnel auth) + std::string proxy_username; + std::string proxy_password; // no_proxy StringList no_proxy_hosts; //private: @@ -45,6 +50,7 @@ struct http_client_s { hssl_ctx_t ssl_ctx; bool alloced_ssl_ctx; HttpParserPtr parser; + HttpConnKey conn_key; // for async std::mutex mutex_; std::shared_ptr async_client_; @@ -166,6 +172,12 @@ int http_client_add_no_proxy(http_client_t* cli, const char* host) { return 0; } +int http_client_set_proxy_auth(http_client_t* cli, const char* username, const char* password) { + cli->proxy_username = username ? username : ""; + cli->proxy_password = password ? password : ""; + return 0; +} + static int http_client_make_request(http_client_t* cli, HttpRequest* req) { if (req->url.empty() || *req->url.c_str() == '/') { req->scheme = cli->https ? "https" : "http"; @@ -174,24 +186,37 @@ static int http_client_make_request(http_client_t* cli, HttpRequest* req) { } req->ParseUrl(); - bool https = req->IsHttps(); - bool use_proxy = https ? (!cli->https_proxy_host.empty()) : (!cli->http_proxy_host.empty()); - if (use_proxy) { - if (req->host == "127.0.0.1" || req->host == "localhost") { - use_proxy = false; - } - } - if (use_proxy) { - for (const auto& host : cli->no_proxy_hosts) { - if (req->host == host) { + if (!req->IsTunnelProxy() && !req->IsProxy()) { + bool https = req->IsHttps(); + bool use_proxy = https ? (!cli->https_proxy_host.empty()) : (!cli->http_proxy_host.empty()); + if (use_proxy) { + if (req->host == "127.0.0.1" || req->host == "localhost") { use_proxy = false; - break; + } + } + if (use_proxy) { + for (const auto& host : cli->no_proxy_hosts) { + if (req->host == host) { + use_proxy = false; + break; + } + } + } + + if (use_proxy) { + if (https) { + req->SetTunnelProxy(cli->https_proxy_host.c_str(), cli->https_proxy_port, + cli->proxy_username.empty() ? NULL : cli->proxy_username.c_str(), + cli->proxy_password.empty() ? NULL : cli->proxy_password.c_str()); + } else { + req->SetProxy(cli->http_proxy_host.c_str(), cli->http_proxy_port); + req->SetProxyAuth(cli->proxy_username.c_str(), cli->proxy_password.c_str()); } } } - if (use_proxy) { - req->SetProxy(https ? cli->https_proxy_host.c_str() : cli->http_proxy_host.c_str(), - https ? cli->https_proxy_port : cli->http_proxy_port); + + if (!req->IsProxy()) { + req->SetProxyAuth(NULL); } if (req->timeout == 0) { @@ -207,79 +232,208 @@ static int http_client_make_request(http_client_t* cli, HttpRequest* req) { return 0; } -int http_client_connect(http_client_t* cli, const char* host, int port, int https, int timeout) { +static void http_client_copy_settings(http_client_t* dst, const http_client_t* src) { + dst->host = src->host; + dst->port = src->port; + dst->https = src->https; + dst->timeout = src->timeout; + dst->headers = src->headers; + dst->http_proxy_host = src->http_proxy_host; + dst->http_proxy_port = src->http_proxy_port; + dst->https_proxy_host = src->https_proxy_host; + dst->https_proxy_port = src->https_proxy_port; + dst->proxy_username = src->proxy_username; + dst->proxy_password = src->proxy_password; + dst->no_proxy_hosts = src->no_proxy_hosts; + dst->ssl_ctx = src->ssl_ctx; + dst->alloced_ssl_ctx = false; +} + +// Client-side TLS handshake on an already-connected fd. sni_host is the origin +// host used for SNI. On failure frees cli->ssl and returns an error (<0); the +// caller owns/closes connfd. +static int http_client_ssl_handshake(http_client_t* cli, int connfd, const char* sni_host, + int blocktime, unsigned int start_time) { + // cli->ssl_ctx > g_ssl_ctx > hssl_ctx_new + hssl_ctx_t ssl_ctx = NULL; + if (cli->ssl_ctx) { + ssl_ctx = cli->ssl_ctx; + } else if (g_ssl_ctx) { + ssl_ctx = g_ssl_ctx; + } else { + cli->ssl_ctx = ssl_ctx = hssl_ctx_new(NULL); + cli->alloced_ssl_ctx = true; + } + if (ssl_ctx == NULL) { + return NABS(ERR_NEW_SSL_CTX); + } + cli->ssl = hssl_new(ssl_ctx, connfd); + if (cli->ssl == NULL) { + return NABS(ERR_NEW_SSL); + } + if (sni_host && !is_ipaddr(sni_host)) { + hssl_set_sni_hostname(cli->ssl, sni_host); + } +#ifdef WITH_OPENSSL + // Offer ALPN "h2" only when HTTP/2 is intended, so an h2-capable server + // negotiates it (real https servers require ALPN, not prior-knowledge). + // Set it per-connection on the SSL object (not the shared ctx), so it + // works with any ctx source (user/global/allocated) and never leaks h2 + // into other clients or http/1.1 requests. + if (cli->http_version == 2) { + static unsigned char s_alpn_protos[] = "\x02h2\x08http/1.1"; + hssl_set_alpn_protos(cli->ssl, s_alpn_protos, sizeof(s_alpn_protos) - 1); + } +#endif + unsigned int elapsed = gettick_ms() - start_time; + int ssl_timeout = blocktime - (int)elapsed; + if (ssl_timeout <= 0) { + hssl_free(cli->ssl); + cli->ssl = NULL; + return NABS(ETIMEDOUT); + } + so_rcvtimeo(connfd, ssl_timeout); + int ret = hssl_connect(cli->ssl); + if (ret != 0) { + fprintf(stderr, "* ssl handshake failed: %d\n", ret); + hloge("ssl handshake failed: %d", ret); + hssl_free(cli->ssl); + cli->ssl = NULL; + return NABS(ret); + } + return 0; +} + +// Blocking HTTP CONNECT handshake to a proxy on connfd: sends +// CONNECT origin_host:origin_port HTTP/1.1 ... [Proxy-Authorization] \r\n\r\n +// then reads the response headers and requires a 2xx status. Returns 0 on +// success; the caller owns/closes connfd. +static int http_client_http_connect(int connfd, const char* origin_host, int origin_port, + const std::string& user, const std::string& pass, + int blocktime) { + // authority form; bracket IPv6 literals per RFC 3986 + std::string authority = is_ipv6(origin_host) + ? hv::asprintf("[%s]:%d", origin_host, origin_port) + : hv::asprintf("%s:%d", origin_host, origin_port); + std::string reqstr = hv::asprintf("CONNECT %s HTTP/1.1\r\nHost: %s\r\n", + authority.c_str(), authority.c_str()); + if (!user.empty()) { + std::string cred = user + ":" + pass; + reqstr += "Proxy-Authorization: Basic " + + hv::Base64Encode((const unsigned char*)cred.data(), cred.size()) + "\r\n"; + } + reqstr += "\r\n"; + + unsigned int start = gettick_ms(); + size_t total = 0; + while (total < reqstr.size()) { + int left = blocktime - (int)(gettick_ms() - start); + if (left <= 0) return NABS(ETIMEDOUT); + so_sndtimeo(connfd, left); + int nsend = send(connfd, reqstr.data() + total, reqstr.size() - total, 0); + if (nsend <= 0) { + if (socket_errno() == EINTR) continue; + return NABS(socket_errno()); + } + total += nsend; + } + + // read response headers until CRLFCRLF + char buf[1024]; + int rlen = 0; + while (rlen < (int)sizeof(buf)) { + int left = blocktime - (int)(gettick_ms() - start); + if (left <= 0) return NABS(ETIMEDOUT); + so_rcvtimeo(connfd, left); + int nrecv = recv(connfd, buf + rlen, sizeof(buf) - rlen, 0); + if (nrecv == 0) return NABS(ERR_CONNECT); // peer closed + if (nrecv < 0) { + if (socket_errno() == EINTR) continue; + return NABS(socket_errno()); + } + rlen += nrecv; + for (int i = 3; i < rlen; ++i) { + if (buf[i-3]=='\r' && buf[i-2]=='\n' && buf[i-1]=='\r' && buf[i]=='\n') { + int code = 0; + const char* sp = (const char*)memchr(buf, ' ', rlen); + if (sp) code = atoi(sp + 1); + if (code >= 200 && code < 300) return 0; + hloge("http proxy CONNECT failed: %d", code); + return NABS(ERR_CONNECT); + } + } + } + return NABS(ERR_CONNECT); // headers too large, no blank line +} + +int http_client_connect(http_client_t* cli, HttpRequest* req) { + if (cli == NULL || req == NULL) return ERR_NULL_POINTER; + + if (!req->url.empty()) { + req->ParseUrl(); + } cli->Close(); int blocktime = DEFAULT_CONNECT_TIMEOUT; + int timeout = MIN(req->connect_timeout, req->timeout); if (timeout > 0) { blocktime = MIN(timeout*1000, blocktime); } + + bool tunnel = req->IsTunnelProxy(); + bool https = req->IsHttps() && !req->IsProxy(); + const char* connect_host = tunnel ? req->tunnel_proxy_host.c_str() : req->host.c_str(); + int connect_port = tunnel ? req->tunnel_proxy_port : req->port; + unsigned int start_time = gettick_ms(); - int connfd = ConnectTimeout(host, port, blocktime); + int connfd = ConnectTimeout(connect_host, connect_port, blocktime); if (connfd < 0) { - hloge("connect %s:%d failed!", host, port); + hloge("connect %s:%d failed!", connect_host, connect_port); return connfd; } tcp_nodelay(connfd, 1); - if (https && cli->ssl == NULL) { - // cli->ssl_ctx > g_ssl_ctx > hssl_ctx_new - hssl_ctx_t ssl_ctx = NULL; - if (cli->ssl_ctx) { - ssl_ctx = cli->ssl_ctx; - } else if (g_ssl_ctx) { - ssl_ctx = g_ssl_ctx; - } else { - cli->ssl_ctx = ssl_ctx = hssl_ctx_new(NULL); - cli->alloced_ssl_ctx = true; - } - if (ssl_ctx == NULL) { - closesocket(connfd); - return NABS(ERR_NEW_SSL_CTX); - } - cli->ssl = hssl_new(ssl_ctx, connfd); - if (cli->ssl == NULL) { + if (tunnel) { + int left = blocktime - (int)(gettick_ms() - start_time); + if (left <= 0) { closesocket(connfd); - return NABS(ERR_NEW_SSL); - } - if (!is_ipaddr(host)) { - hssl_set_sni_hostname(cli->ssl, host); - } -#ifdef WITH_OPENSSL - // Offer ALPN "h2" only when HTTP/2 is intended, so an h2-capable server - // negotiates it (real https servers require ALPN, not prior-knowledge). - // Set it per-connection on the SSL object (not the shared ctx), so it - // works with any ctx source (user/global/allocated) and never leaks h2 - // into other clients or http/1.1 requests. - if (cli->http_version == 2) { - static unsigned char s_alpn_protos[] = "\x02h2\x08http/1.1"; - hssl_set_alpn_protos(cli->ssl, s_alpn_protos, sizeof(s_alpn_protos) - 1); + return NABS(ETIMEDOUT); } -#endif - unsigned int elapsed = gettick_ms() - start_time; - int ssl_timeout = blocktime - (int)elapsed; - if (ssl_timeout <= 0) { - hssl_free(cli->ssl); - cli->ssl = NULL; + int ret = http_client_http_connect(connfd, req->host.c_str(), req->port, + req->tunnel_proxy_username, req->tunnel_proxy_password, left); + if (ret != 0) { closesocket(connfd); - return NABS(ETIMEDOUT); + return ret; } - so_rcvtimeo(connfd, ssl_timeout); - int ret = hssl_connect(cli->ssl); + } + + cli->http_version = req->http_major; + if (https && cli->ssl == NULL) { + int ret = http_client_ssl_handshake(cli, connfd, req->host.c_str(), blocktime, start_time); if (ret != 0) { - fprintf(stderr, "* ssl handshake failed: %d\n", ret); - hloge("ssl handshake failed: %d", ret); - hssl_free(cli->ssl); - cli->ssl = NULL; closesocket(connfd); - return NABS(ret); + return ret; } } cli->fd = connfd; cli->keepalive_requests = 0; + cli->conn_key = HttpConnKey(*req); return connfd; } +int http_client_connect(http_client_t* cli, const char* host, int port, int https, int timeout) { + if (cli == NULL || host == NULL) return ERR_NULL_POINTER; + + HttpRequest req; + req.scheme = https ? "https" : "http"; + req.host = host; + req.port = port; + if (timeout > 0) { + req.connect_timeout = timeout; + } + return http_client_connect(cli, &req); +} + int http_client_close(http_client_t* cli) { if (cli == NULL) return 0; cli->Close(); @@ -335,26 +489,20 @@ static int http_client_exec(http_client_t* cli, HttpRequest* req, HttpResponse* retry_count = 1; } - if (cli->parser == NULL) { - cli->parser = HttpParserPtr(HttpParser::New(HTTP_CLIENT, (http_version)req->http_major)); - if (cli->parser == NULL) { - hloge("New HttpParser failed!"); - return ERR_NULL_POINTER; - } - } - char recvbuf[1024] = {0}; char* data = NULL; size_t len = 0; int total_nsend, nsend, nrecv; total_nsend = nsend = nrecv = 0; - if (connfd <= 0 || cli->host != req->host || cli->port != req->port) { + HttpConnKey conn_key(*req); + int want_major = req->http_major; + if (connfd <= 0 || cli->conn_key != conn_key) { cli->host = req->host; cli->port = req->port; cli->http_version = req->http_major; // gates the ALPN "h2" offer in connect connect: - connfd = http_client_connect(cli, req->host.c_str(), req->port, https, connect_timeout); + connfd = http_client_connect(cli, req); if (connfd < 0) { return connfd; } @@ -370,22 +518,27 @@ static int http_client_exec(http_client_t* cli, HttpRequest* req, HttpResponse* unsigned int alpn_len = 0; const char* alpn = hssl_get_alpn_proto(cli->ssl, &alpn_len); bool is_h2 = (alpn && alpn_len == 2 && memcmp(alpn, "h2", 2) == 0); - int want_major = is_h2 ? 2 : 1; + want_major = is_h2 ? 2 : 1; if (want_major == 1) { // fall back to HTTP/1.1 (not 1.0: keep-alive semantics) req->http_major = 1; req->http_minor = 1; } - // rebuild the parser if it doesn't match the negotiated version - if (cli->parser == NULL || cli->http_version != want_major) { - cli->parser = HttpParserPtr(HttpParser::New(HTTP_CLIENT, - (http_version)(want_major == 2 ? HTTP_V2 : HTTP_V1))); - cli->http_version = want_major; - } } #endif } + if (cli->parser == NULL || cli->http_version != want_major) { + cli->parser = HttpParserPtr(HttpParser::New(HTTP_CLIENT, + (http_version)(want_major == 2 ? HTTP_V2 : HTTP_V1))); + if (cli->parser == NULL) { + hloge("New HttpParser failed!"); + cli->Close(); + return ERR_NULL_POINTER; + } + cli->http_version = want_major; + } + cli->parser->SubmitRequest(req); send: data = NULL; @@ -574,10 +727,23 @@ static int http_client_exec_curl(http_client_t* cli, HttpRequest* req, HttpRespo } CURL* curl = cli->curl; - // proxy + // proxy: plain-http forward proxy (req->host is the proxy) or CONNECT + // tunnel (tunnel_proxy_* for https). libcurl handles both, incl. CONNECT. + // cli->curl is reused across requests, so always reset proxy options first + // (a stale PROXY / PROXYUSERPWD would otherwise leak into a later request + // that uses a different or no proxy). + curl_easy_setopt(curl, CURLOPT_PROXY, ""); + curl_easy_setopt(curl, CURLOPT_PROXYUSERPWD, ""); if (req->IsProxy()) { curl_easy_setopt(curl, CURLOPT_PROXY, req->host.c_str()); curl_easy_setopt(curl, CURLOPT_PROXYPORT, req->port); + } else if (req->IsTunnelProxy()) { + curl_easy_setopt(curl, CURLOPT_PROXY, req->tunnel_proxy_host.c_str()); + curl_easy_setopt(curl, CURLOPT_PROXYPORT, req->tunnel_proxy_port); + if (!req->tunnel_proxy_username.empty()) { + std::string userpwd = req->tunnel_proxy_username + ":" + req->tunnel_proxy_password; + curl_easy_setopt(curl, CURLOPT_PROXYUSERPWD, userpwd.c_str()); + } } // SSL @@ -709,24 +875,64 @@ const char* http_client_strerror(int errcode) { #endif -static int http_client_redirect(HttpRequest* req, HttpResponse* resp) { +static std::string http_client_redirect_url(const HttpRequest* req, const std::string& location) { + if (location.empty() || strstr(location.c_str(), "://") != NULL) { + return location; + } + if (location[0] != '/') { + return location; + } + + HUrl base; + if (!base.parse(req->url) || base.scheme.empty() || base.host.empty()) { + return location; + } + return base.scheme + "://" + hv::NetAddr::to_string(base.host.c_str(), base.port) + location; +} + +static void http_client_prepare_redirect(HttpRequest* req, const std::string& location) { + bool old_https = req->IsHttps(); + std::string redirect_url = http_client_redirect_url(req, location); + bool new_https = hv::startswith(redirect_url, "https://") ? true : + hv::startswith(redirect_url, "http://") ? false : old_https; + + if (new_https && req->IsProxy()) { + req->proxy = 0; + } else if (!new_https && req->IsTunnelProxy()) { + req->tunnel_proxy_host.clear(); + req->tunnel_proxy_port = 0; + req->tunnel_proxy_username.clear(); + req->tunnel_proxy_password.clear(); + } + + req->url = redirect_url; + req->headers.erase("Host"); + req->ParseUrl(); +} + +static int http_client_send_impl(http_client_t* cli, HttpRequest* req, HttpResponse* resp, bool preserve_connection); + +static int http_client_redirect(http_client_t* cli, HttpRequest* req, HttpResponse* resp, bool preserve_connection) { std::string location = resp->headers["Location"]; if (!location.empty()) { hlogi("redirect %s => %s", req->url.c_str(), location.c_str()); - req->url = location; - req->ParseUrl(); - req->headers["Host"] = req->host; + http_client_prepare_redirect(req, location); + http_client_make_request(cli, req); resp->Reset(); - return http_client_send(req, resp); + + HttpConnKey redirect_key(*req); + if (!preserve_connection || cli->fd < 0 || cli->conn_key == redirect_key) { + return http_client_send_impl(cli, req, resp, preserve_connection); + } + + http_client_t redirect_cli; + http_client_copy_settings(&redirect_cli, cli); + return http_client_send_impl(&redirect_cli, req, resp, false); } return 0; } -int http_client_send(http_client_t* cli, HttpRequest* req, HttpResponse* resp) { - if (!cli || !req || !resp) return ERR_NULL_POINTER; - - http_client_make_request(cli, req); - +static int http_client_send_impl(http_client_t* cli, HttpRequest* req, HttpResponse* resp, bool preserve_connection) { if (req->http_cb) resp->http_cb = std::move(req->http_cb); #if WITH_CURL @@ -738,11 +944,18 @@ int http_client_send(http_client_t* cli, HttpRequest* req, HttpResponse* resp) { // redirect if (req->redirect && HTTP_STATUS_IS_REDIRECT(resp->status_code)) { - return http_client_redirect(req, resp); + return http_client_redirect(cli, req, resp, preserve_connection); } return 0; } +int http_client_send(http_client_t* cli, HttpRequest* req, HttpResponse* resp) { + if (!cli || !req || !resp) return ERR_NULL_POINTER; + + http_client_make_request(cli, req); + return http_client_send_impl(cli, req, resp, true); +} + int http_client_send(HttpRequest* req, HttpResponse* resp) { if (!req || !resp) return ERR_NULL_POINTER; diff --git a/http/client/HttpClient.h b/http/client/HttpClient.h index 984096d59..0fc9fe760 100644 --- a/http/client/HttpClient.h +++ b/http/client/HttpClient.h @@ -53,6 +53,8 @@ HV_EXPORT int http_client_set_http_proxy(http_client_t* cli, const char* host, i HV_EXPORT int http_client_set_https_proxy(http_client_t* cli, const char* host, int port); // no_proxy HV_EXPORT int http_client_add_no_proxy(http_client_t* cli, const char* host); +// proxy auth (Basic for http forward proxy / HTTP CONNECT tunnel) +HV_EXPORT int http_client_set_proxy_auth(http_client_t* cli, const char* username, const char* password); // sync HV_EXPORT int http_client_send(http_client_t* cli, HttpRequest* req, HttpResponse* resp); @@ -70,6 +72,8 @@ HV_EXPORT int http_client_send_async(HttpRequestPtr req, HttpResponseCallback re // low-level api // @retval >=0 connfd, <0 error +// Connect according to req route: direct or HTTP CONNECT tunnel. +HV_EXPORT int http_client_connect(http_client_t* cli, HttpRequest* req); HV_EXPORT int http_client_connect(http_client_t* cli, const char* host, int port, int https, int timeout); HV_EXPORT int http_client_send_header(http_client_t* cli, HttpRequest* req); HV_EXPORT int http_client_send_data(http_client_t* cli, const char* data, int size); @@ -124,6 +128,10 @@ class HttpClient { int addNoProxy(const char* host) { return http_client_add_no_proxy(client_.get(), host); } + // proxy auth (Basic for http forward proxy / HTTP CONNECT tunnel) + int setProxyAuth(const char* username, const char* password) { + return http_client_set_proxy_auth(client_.get(), username, password); + } // sync int send(HttpRequest* req, HttpResponse* resp) { @@ -136,6 +144,9 @@ class HttpClient { } // low-level api + int connect(HttpRequest& req) { + return http_client_connect(client_.get(), &req); + } int connect(const char* host, int port = DEFAULT_HTTP_PORT, int https = 0, int timeout = DEFAULT_HTTP_CONNECT_TIMEOUT) { return http_client_connect(client_.get(), host, port, https, timeout); }