vinyl-cache/bin/vinyld/http1/cache_http1_line.c
0
/*-
1
 * Copyright (c) 2006 Verdens Gang AS
2
 * Copyright (c) 2006-2011 Varnish Software AS
3
 * All rights reserved.
4
 *
5
 * Author: Poul-Henning Kamp <phk@phk.freebsd.dk>
6
 *
7
 * SPDX-License-Identifier: BSD-2-Clause
8
 *
9
 * Redistribution and use in source and binary forms, with or without
10
 * modification, are permitted provided that the following conditions
11
 * are met:
12
 * 1. Redistributions of source code must retain the above copyright
13
 *    notice, this list of conditions and the following disclaimer.
14
 * 2. Redistributions in binary form must reproduce the above copyright
15
 *    notice, this list of conditions and the following disclaimer in the
16
 *    documentation and/or other materials provided with the distribution.
17
 *
18
 * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
19
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
20
 * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
21
 * ARE DISCLAIMED.  IN NO EVENT SHALL AUTHOR OR CONTRIBUTORS BE LIABLE
22
 * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
23
 * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
24
 * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
25
 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
26
 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
27
 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
28
 * SUCH DAMAGE.
29
 *
30
 * Write data to fd
31
 * We try to use writev() if possible in order to minimize number of
32
 * syscalls made and packets sent.  It also just might allow the worker
33
 * thread to complete the request without holding stuff locked.
34
 *
35
 * XXX: chunked header (generated in Flush) and Tail (EndChunk)
36
 *      are not accounted by means of the size_t returned. Obvious ideas:
37
 *      - add size_t return value to Flush and EndChunk
38
 *      - base accounting on (struct v1l).cnt
39
 */
40
41
#include "config.h"
42
43
#include <sys/uio.h>
44
#include "cache/cache_int.h"
45
#include "cache/cache_filter.h"
46
47
#include <stdio.h>
48
49
#include "cache_http1.h"
50
#include "vtim.h"
51
52
/*--------------------------------------------------------------------*/
53
54
struct v1l {
55
        unsigned                magic;
56
#define V1L_MAGIC               0x2f2142e5
57
        int                     *wfd;
58
        stream_close_t          werr;   /* valid after V1L_Flush() */
59
        struct iovec            *iov;
60
        int                     siov;
61
        int                     niov;
62
        size_t                  liov;
63
        size_t                  cliov;
64
        int                     ciov;   /* Chunked header marker */
65
        vtim_real               deadline;
66
        struct vsl_log          *vsl;
67
        uint64_t                cnt;    /* Flushed byte count */
68
        struct ws               *ws;
69
        uintptr_t               ws_snap;
70
        void                    **vdp_priv;
71
};
72
73
/*--------------------------------------------------------------------
74
 * for niov == 0, reserve the ws for max number of iovs
75
 * otherwise, up to niov
76
 */
77
78
struct v1l *
79 120549
V1L_Open(struct ws *ws, int *fd, struct vsl_log *vsl,
80
    vtim_real deadline, unsigned niov)
81
{
82
        struct v1l *v1l;
83
        unsigned u;
84
        uintptr_t ws_snap;
85
        size_t sz;
86
87 120549
        if (WS_Overflowed(ws))
88 0
                return (NULL);
89
90 120549
        if (niov != 0)
91 71830
                assert(niov >= 3);
92
93 120549
        ws_snap = WS_Snapshot(ws);
94
95 120549
        v1l = WS_Alloc(ws, sizeof *v1l);
96 120549
        if (v1l == NULL)
97 21
                return (NULL);
98 120528
        INIT_OBJ(v1l, V1L_MAGIC);
99
100 120528
        v1l->ws = ws;
101 120528
        v1l->ws_snap = ws_snap;
102
103 120528
        u = WS_ReserveLumps(ws, sizeof(struct iovec));
104 120528
        if (u < 3) {
105
                /* Must have at least 3 in case of chunked encoding */
106 0
                WS_Release(ws, 0);
107 0
                WS_MarkOverflow(ws);
108 0
                return (NULL);
109
        }
110 120528
        if (u > IOV_MAX)
111 1511
                u = IOV_MAX;
112 120528
        if (niov != 0 && u > niov)
113 70237
                u = niov;
114 120528
        v1l->iov = WS_Reservation(ws);
115 120528
        v1l->siov = (int)u;
116 120528
        v1l->ciov = (int)u;
117 120528
        v1l->wfd = fd;
118 120528
        v1l->deadline = deadline;
119 120528
        v1l->vsl = vsl;
120 120528
        v1l->werr = SC_NULL;
121
122 120528
        sz = u * sizeof(struct iovec);
123 120528
        assert(sz < UINT_MAX);
124 120528
        WS_Release(ws, (unsigned)sz);
125 120528
        return (v1l);
126 120549
}
127
128
void
129 1512
V1L_NoRollback(struct v1l *v1l)
130
{
131
132 1512
        CHECK_OBJ_NOTNULL(v1l, V1L_MAGIC);
133 1512
        v1l->ws_snap = 0;
134 1512
}
135
136
stream_close_t
137 120534
V1L_Close(struct v1l **v1lp, uint64_t *cnt)
138
{
139
        struct v1l *v1l;
140
        struct ws *ws;
141
        uintptr_t ws_snap;
142
        stream_close_t sc;
143
144 120534
        AN(cnt);
145 120534
        TAKE_OBJ_NOTNULL(v1l, v1lp, V1L_MAGIC);
146 120534
        if (v1l->vdp_priv != NULL) {
147 91843
                assert(*v1l->vdp_priv == v1l);
148 91843
                *v1l->vdp_priv = NULL;
149 91843
        }
150 120534
        sc = V1L_Flush(v1l);
151 120534
        *cnt = v1l->cnt;
152 120534
        ws = v1l->ws;
153 120534
        ws_snap = v1l->ws_snap;
154 120534
        ZERO_OBJ(v1l, sizeof *v1l);
155 120534
        if (ws_snap != 0)
156 119022
                WS_Rollback(ws, ws_snap);
157 120534
        return (sc);
158
}
159
160
static void
161 368
v1l_prune(struct v1l *v1l, ssize_t abytes)
162
{
163 368
        size_t used = 0;
164
        size_t sz, bytes, used_here;
165
        int j;
166
167 368
        assert(abytes > 0);
168 368
        bytes = (size_t)abytes;
169
170 1884
        for (j = 0; j < v1l->niov; j++) {
171 1884
                if (used + v1l->iov[j].iov_len > bytes) {
172
                        /* Cutoff is in this iov */
173 368
                        used_here = bytes - used;
174 368
                        v1l->iov[j].iov_len -= used_here;
175 368
                        v1l->iov[j].iov_base =
176 368
                            (char*)v1l->iov[j].iov_base + used_here;
177 368
                        sz = (unsigned)v1l->niov - (unsigned)j;
178 368
                        sz *= sizeof(struct iovec);
179 368
                        vmemmove(v1l->iov, &v1l->iov[j], sz);
180 368
                        v1l->niov -= j;
181 368
                        assert(v1l->liov >= bytes);
182 368
                        v1l->liov -= bytes;
183 368
                        return;
184
                }
185 1516
                used += v1l->iov[j].iov_len;
186 1516
        }
187 0
        AZ(v1l->liov);
188 368
}
189
190
stream_close_t
191 527201
V1L_Flush(struct v1l *v1l)
192
{
193
        ssize_t i;
194
        size_t sz;
195
        int err;
196
        char cbuf[32];
197
198 527201
        CHECK_OBJ_NOTNULL(v1l, V1L_MAGIC);
199 527201
        CHECK_OBJ_NOTNULL(v1l->werr, STREAM_CLOSE_MAGIC);
200 527201
        AN(v1l->wfd);
201
202 527201
        assert(v1l->niov <= v1l->siov);
203
204 527201
        if (*v1l->wfd >= 0 && v1l->liov > 0 && v1l->werr == SC_NULL) {
205 447016
                if (v1l->ciov < v1l->siov && v1l->cliov > 0) {
206
                        /* Add chunk head & tail */
207 38592
                        bprintf(cbuf, "00%zx\r\n", v1l->cliov);
208 38592
                        sz = vstrlen(cbuf);
209 38592
                        v1l->iov[v1l->ciov].iov_base = cbuf;
210 38592
                        v1l->iov[v1l->ciov].iov_len = sz;
211 38592
                        v1l->liov += sz;
212
213
                        /* This is OK, because siov was --'ed */
214 38592
                        v1l->iov[v1l->niov].iov_base = cbuf + sz - 2;
215 38592
                        v1l->iov[v1l->niov++].iov_len = 2;
216 38592
                        v1l->liov += 2;
217 447016
                } else if (v1l->ciov < v1l->siov) {
218 1524
                        v1l->iov[v1l->ciov].iov_base = cbuf;
219 1524
                        v1l->iov[v1l->ciov].iov_len = 0;
220 1524
                }
221
222 447016
                i = 0;
223 447016
                err = 0;
224 447016
                do {
225 447926
                        if (VTIM_real() > v1l->deadline) {
226 168
                                VSLb(v1l->vsl, SLT_Debug,
227
                                    "Hit total send timeout, "
228
                                    "wrote = %zd/%zd; not retrying",
229 84
                                    i, v1l->liov);
230 84
                                i = -1;
231 84
                                break;
232
                        }
233
234 447842
                        i = writev(*v1l->wfd, v1l->iov, v1l->niov);
235 447842
                        if (i > 0) {
236 447163
                                v1l->cnt += (size_t)i;
237 447163
                                if ((size_t)i == v1l->liov)
238 446795
                                        break;
239 368
                        }
240
241
                        /* we hit a timeout, and some data may have been sent:
242
                         * Remove sent data from start of I/O vector, then retry
243
                         *
244
                         * XXX: Add a "minimum sent data per timeout counter to
245
                         * prevent slowloris attacks
246
                         */
247
248 1047
                        err = errno;
249
250 1047
                        if (err == EWOULDBLOCK) {
251 1064
                                VSLb(v1l->vsl, SLT_Debug,
252
                                    "Hit idle send timeout, "
253
                                    "wrote = %zd/%zd; retrying",
254 532
                                    i, v1l->liov);
255 532
                        }
256
257 1047
                        if (i > 0)
258 368
                                v1l_prune(v1l, i);
259 1047
                } while (i > 0 || err == EWOULDBLOCK);
260
261 447016
                if (i <= 0) {
262 462
                        VSLb(v1l->vsl, SLT_Debug,
263
                            "Write error, retval = %zd, len = %zd, errno = %s",
264 231
                            i, v1l->liov, VAS_errtxt(err));
265 231
                        assert(v1l->werr == SC_NULL);
266 231
                        if (err == EPIPE)
267 146
                                v1l->werr = SC_REM_CLOSE;
268
                        else
269 85
                                v1l->werr = SC_TX_ERROR;
270 231
                        errno = err;
271 231
                }
272 447016
        }
273 527207
        v1l->liov = 0;
274 527207
        v1l->cliov = 0;
275 527207
        v1l->niov = 0;
276 527207
        if (v1l->ciov < v1l->siov)
277 63136
                v1l->ciov = v1l->niov++;
278 527185
        CHECK_OBJ_NOTNULL(v1l->werr, STREAM_CLOSE_MAGIC);
279 527185
        return (v1l->werr);
280
}
281
282
size_t
283 3094159
V1L_Write(struct v1l *v1l, const void *ptr, ssize_t alen)
284
{
285 3094159
        size_t len = 0;
286
287 3094159
        CHECK_OBJ_NOTNULL(v1l, V1L_MAGIC);
288 3094159
        AN(v1l->wfd);
289 3094159
        if (alen == 0 || *v1l->wfd < 0)
290 752
                return (0);
291 3094159
        if (alen > 0)
292 1723594
                len = (size_t)alen;
293 1370565
        else if (alen == -1)
294 1370565
                len = vstrlen(ptr);
295
        else
296 0
                WRONG("alen");
297
298 3094159
        assert(v1l->niov < v1l->siov);
299 3094159
        v1l->iov[v1l->niov].iov_base = TRUST_ME(ptr);
300 3094159
        v1l->iov[v1l->niov].iov_len = len;
301 3094159
        v1l->liov += len;
302 3094159
        v1l->niov++;
303 3094159
        v1l->cliov += len;
304 3094159
        if (v1l->niov >= v1l->siov) {
305 2373
                (void)V1L_Flush(v1l);
306 2373
                VSC_C_main->http1_iovs_flush++;
307 2373
        }
308 3094159
        return (len);
309 3094159
}
310
311
void
312 8061
V1L_Chunked(struct v1l *v1l)
313
{
314
315 8061
        CHECK_OBJ_NOTNULL(v1l, V1L_MAGIC);
316
317 8061
        assert(v1l->ciov == v1l->siov);
318 8061
        assert(v1l->siov >= 3);
319
        /*
320
         * If there is no space for chunked header, a chunk of data and
321
         * a chunk tail, we might as well flush right away.
322
         */
323 8061
        if (v1l->niov + 3 >= v1l->siov) {
324 0
                (void)V1L_Flush(v1l);
325 0
                VSC_C_main->http1_iovs_flush++;
326 0
        }
327 8061
        v1l->siov--;
328 8061
        v1l->ciov = v1l->niov++;
329 8061
        v1l->cliov = 0;
330 8061
        assert(v1l->ciov < v1l->siov);
331 8061
        assert(v1l->niov < v1l->siov);
332 8061
}
333
334
/*
335
 * XXX: It is not worth the complexity to attempt to get the
336
 * XXX: end of chunk into the V1L_Flush(), because most of the time
337
 * XXX: if not always, that is a no-op anyway, because the calling
338
 * XXX: code already called V1L_Flush() to release local storage.
339
 */
340
341
void
342 7601
V1L_EndChunk(struct v1l *v1l)
343
{
344
345 7601
        CHECK_OBJ_NOTNULL(v1l, V1L_MAGIC);
346
347 7601
        assert(v1l->ciov < v1l->siov);
348 7601
        (void)V1L_Flush(v1l);
349 7601
        v1l->siov++;
350 7601
        v1l->ciov = v1l->siov;
351 7601
        v1l->niov = 0;
352 7601
        v1l->cliov = 0;
353 7601
        (void)V1L_Write(v1l, "0\r\n\r\n", -1);
354 7601
}
355
356
/*--------------------------------------------------------------------
357
 * VDP using V1L
358
 */
359
360
/* remember priv pointer for V1L_Close() to clear */
361
static int v_matchproto_(vdp_init_f)
362 91168
v1l_init(VRT_CTX, struct vdp_ctx *vdc, void **priv)
363
{
364
        struct v1l *v1l;
365
366 91168
        (void) ctx;
367 91168
        (void) vdc;
368 91168
        AN(priv);
369 91168
        CAST_OBJ_NOTNULL(v1l, *priv, V1L_MAGIC);
370
371 91168
        v1l->vdp_priv = priv;
372 91168
        return (0);
373
}
374
375
static int v_matchproto_(vdp_bytes_f)
376 507282
v1l_bytes(struct vdp_ctx *vdc, enum vdp_action act, void **priv,
377
    const void *ptr, ssize_t len)
378
{
379 507282
        size_t wl = 0;
380
381 507282
        CHECK_OBJ_NOTNULL(vdc, VDP_CTX_MAGIC);
382 507282
        AN(priv);
383
384 507282
        AZ(vdc->nxt);           /* always at the bottom of the pile */
385
386 507282
        if (len > 0)
387 477224
                wl = V1L_Write(*priv, ptr, len);
388 507282
        if (act > VDP_NULL && V1L_Flush(*priv) != SC_NULL)
389 229
                return (-1);
390 507053
        if ((size_t)len != wl)
391 0
                return (-1);
392 507053
        return (0);
393 507282
}
394
395
/*--------------------------------------------------------------------
396
 * VDPIO using V1L
397
 *
398
 * this is deliverately half-baked to reduce work in progress while heading
399
 * towards VAI/VDPIO: we update the v1l with the scarab, which we
400
 * return unmodified.
401
 *
402
 */
403
404
/* remember priv pointer for V1L_Close() to clear */
405
static int v_matchproto_(vpio_init_f)
406 672
v1l_io_init(VRT_CTX, struct vdp_ctx *vdc, void **priv, int capacity)
407
{
408
        struct v1l *v1l;
409
410 672
        (void) ctx;
411 672
        (void) vdc;
412 672
        AN(priv);
413
414 672
        CAST_OBJ_NOTNULL(v1l, *priv, V1L_MAGIC);
415
416 672
        v1l->vdp_priv = priv;
417 672
        return (capacity);
418
}
419
420
static int v_matchproto_(vpio_init_f)
421 0
v1l_io_upgrade(VRT_CTX, struct vdp_ctx *vdc, void **priv, int capacity)
422
{
423 0
        return (v1l_io_init(ctx, vdc, priv, capacity));
424
}
425
426
/*
427
 * API note
428
 *
429
 * this VDP is special in that it does not transform data, but prepares
430
 * the write. From the perspective of VDPIO, its current state is only
431
 * transitional.
432
 *
433
 * Because the VDP prepares the actual writes, but the caller needs
434
 * to return the scarab's leases, the caller in this case is
435
 * required to empty the scarab after V1L_Flush()'ing.
436
 */
437
438
static int v_matchproto_(vdpio_lease_f)
439 4031
v1l_io_lease(struct vdp_ctx *vdc, struct vdp_entry *this, struct vscarab *scarab)
440
{
441
        struct v1l *v1l;
442
        struct viov *v;
443
        int r;
444
445 4031
        CHECK_OBJ_NOTNULL(vdc, VDP_CTX_MAGIC);
446 4031
        CHECK_OBJ_NOTNULL(this, VDP_ENTRY_MAGIC);
447 4031
        CAST_OBJ_NOTNULL(v1l, this->priv, V1L_MAGIC);
448 4031
        VSCARAB_CHECK(scarab);
449 4031
        AZ(scarab->used);       // see note above
450 4031
        this->calls++;
451 4031
        r = vdpio_pull(vdc, this, scarab);
452 4031
        if (r < 0)
453 2687
                return (r);
454 2688
        VSCARAB_FOREACH(v, scarab)
455 1344
                this->bytes_in += V1L_Write(v1l, v->iov.iov_base, v->iov.iov_len);
456 1344
        return (r);
457 4031
}
458
459
const struct vdp * const VDP_v1l = &(struct vdp){
460
        .name =         "V1B",
461
        .init =         v1l_init,
462
        .bytes =        v1l_bytes,
463
464
        .io_init =      v1l_io_init,
465
        .io_upgrade =   v1l_io_upgrade,
466
        .io_lease =     v1l_io_lease,
467
};