]> git.ipfire.org Git - thirdparty/libarchive.git/commitdiff
warc: modernize reader comments
authordatauwu <datauwu@users.noreply.github.com>
Sat, 11 Jul 2026 08:47:23 +0000 (16:47 +0800)
committerdatauwu <datauwu@users.noreply.github.com>
Sat, 11 Jul 2026 08:47:23 +0000 (16:47 +0800)
Rewrite WARC reader comments to use clearer wording and match the
style used by other libarchive format readers.

Keep the change comment-only while preserving existing behavior.

libarchive/archive_read_support_format_warc.c

index 6790e2fc8f9e4b8177267e5714fa04589fd99aca..6340751175ba0100bd3e297720ae3557f038a44c 100644 (file)
 
 #include "archive_platform.h"
 
-/**
- * WARC is standardised by ISO TC46/SC4/WG12 and currently available as
- * ISO 28500:2009.
- * For the purposes of this file we used the final draft from:
+/*
+ * An overview of WARC format:
+ *
+ * WARC files are laid out as a sequence of records.  Each record has
+ * a text header followed by a content block whose size is given by
+ * Content-Length.  This reader supports WARC/0.12 through WARC/1.0
+ * and was written using the final draft that became ISO 28500:2009:
  * http://bibnum.bnf.fr/warc/WARC_ISO_28500_version1_latestdraft.pdf
  *
- * Todo:
- * [ ] real-world warcs can contain resources at endpoints ending in /
- *     e.g. http://bibnum.bnf.fr/warc/
- *     if you're lucky their response contains a Content-Location: header
- *     pointing to a unix-compliant filename, in the example above it's
- *     Content-Location: http://bibnum.bnf.fr/warc/index.html
- *     however, that's not mandated and github for example doesn't follow
- *     this convention.
- *     We need a set of archive options to control what to do with
- *     entries like these, at the moment care is taken to skip them.
+ * This reader exposes resource and response records as regular files
+ * when they have a usable WARC-Target-URI.  WARC-Date is exposed as
+ * ctime, and a Last-Modified record header is exposed as mtime when
+ * present.
  *
- **/
+ * TODO: Real-world WARCs can contain resources at endpoints ending in
+ * a slash, for example http://bibnum.bnf.fr/warc/.  Some responses
+ * include a Content-Location header that points to a Unix-compatible
+ * filename such as http://bibnum.bnf.fr/warc/index.html, but WARC does
+ * not require that convention and some sites do not follow it.  Until
+ * archive options exist to control these entries, this reader skips
+ * them instead of creating directory endpoints as files.
+ */
 
 #ifdef HAVE_SYS_STAT_H
 #include <sys/stat.h>
 
 typedef enum {
        WT_NONE,
-       /* warcinfo */
+       /* WARC info */
        WT_INFO,
-       /* metadata */
+       /* Metadata */
        WT_META,
-       /* resource */
+       /* Resource */
        WT_RSRC,
-       /* request, unsupported */
+       /* Request, unsupported */
        WT_REQ,
-       /* response, unsupported */
+       /* Response */
        WT_RSP,
-       /* revisit, unsupported */
+       /* Revisit, unsupported */
        WT_RVIS,
-       /* conversion, unsupported */
+       /* Conversion, unsupported */
        WT_CONV,
-       /* continuation, unsupported at the moment */
+       /* Continuation, currently unsupported */
        WT_CONT,
-       /* invalid type */
+       /* Invalid type */
        LAST_WT
 } warc_type_t;
 
@@ -105,18 +109,18 @@ typedef struct {
 } warc_strbuf_t;
 
 struct warc_s {
-       /* content length ahead */
+       /* Content length of the current record */
        int64_t cntlen;
-       /* and how much we've processed so far */
+       /* Bytes processed from the current record */
        int64_t cntoff;
-       /* and how much we need to consume between calls */
+       /* Bytes to consume before the next read */
        size_t unconsumed;
 
-       /* string pool */
+       /* String pool */
        warc_strbuf_t pool;
-       /* previous version */
+       /* Previous version */
        unsigned int pver;
-       /* stringified format name */
+       /* Stringified format name */
        struct archive_string sver;
 };
 
@@ -126,7 +130,7 @@ static int _warc_read(struct archive_read*, const void**, size_t*, int64_t*);
 static int _warc_skip(struct archive_read *a);
 static int _warc_rdhdr(struct archive_read *a, struct archive_entry *e);
 
-/* private routines */
+/* Private routines */
 static unsigned int _warc_rdver(const char *buf, size_t bsz);
 static unsigned int _warc_rdtyp(const char *buf, size_t bsz);
 static warc_string_t _warc_rduri(const char *buf, size_t bsz);
@@ -187,23 +191,23 @@ _warc_bid(struct archive_read *a, int best_bid)
 
        (void)best_bid; /* UNUSED */
 
-       /* check first line of file, it should be a record already */
+       /* Check the first line, which should already be a record header. */
        if ((hdr = __archive_read_ahead(a, 12U, &nrd)) == NULL) {
-               /* no idea what to do */
+               /* Not enough data to identify this format. */
                return -1;
        } else if (nrd < 12) {
-               /* nah, not for us, our magic cookie is at least 12 bytes */
+               /* Not WARC; the magic string is at least 12 bytes. */
                return -1;
        }
 
-       /* otherwise snarf the record's version number */
+       /* Parse the record version number. */
        ver = _warc_rdver(hdr, nrd);
        if (ver < 1200U || ver > 10000U) {
-               /* we only support WARC 0.12 to 1.0 */
+               /* Only WARC 0.12 through WARC 1.0 are supported. */
                return -1;
        }
 
-       /* otherwise be confident */
+       /* WARC magic and version checks passed. */
        return (64);
 }
 
@@ -217,25 +221,24 @@ _warc_rdhdr(struct archive_read *a, struct archive_entry *entry)
        ssize_t nrd;
        const char *eoh;
        char *tmp;
-       /* for the file name, saves some strndup()'ing */
+       /* Reuse the header buffer while parsing the file name. */
        warc_string_t fnam;
-       /* warc record type, not that we really use it a lot */
+       /* WARC record type */
        warc_type_t ftyp;
-       /* content-length+error monad */
+       /* Content length, or a negative error indicator */
        int64_t cntlen;
-       /* record time is the WARC-Date time we reinterpret it as ctime */
+       /* WARC-Date is exposed as the entry ctime. */
        time_t rtime;
-       /* mtime is the Last-Modified time which will be the entry's mtime */
+       /* A Last-Modified record header is exposed as the entry mtime. */
        time_t mtime;
 
 start_over:
-       /* just use read_ahead() they keep track of unconsumed
-        * bits and bobs for us; no need to put an extra shift in
-        * and reproduce that functionality here */
+       /* Use read_ahead(); it already tracks unconsumed bytes, so this
+        * reader does not need a separate shift buffer. */
        buf = __archive_read_ahead(a, HDR_PROBE_LEN, &nrd);
 
        if (nrd < 0) {
-               /* no good */
+               /* I/O or stream error. */
                archive_set_error(
                        &a->archive, ARCHIVE_ERRNO_MISC,
                        "Bad record header");
@@ -245,19 +248,17 @@ start_over:
                 * must be EOF therefore */
                return (ARCHIVE_EOF);
        }
-       /* looks good so far, try and find the end of the header now */
+       /* Locate the end of the record header. */
        eoh = _warc_find_eoh(buf, nrd);
        if (eoh == NULL) {
-               /* still no good, the header end might be beyond the
-                * probe we've requested, but then again who'd cram
-                * so much stuff into the header *and* be 28500-compliant */
+               /* The header terminator was not found in the probed data. */
                archive_set_error(
                        &a->archive, ARCHIVE_ERRNO_MISC,
                        "Bad record header");
                return (ARCHIVE_FATAL);
        }
        ver = _warc_rdver(buf, eoh - buf);
-       /* we currently support WARC 0.12 to 1.0 */
+       /* Only WARC 0.12 through WARC 1.0 are supported. */
        if (ver == 0U) {
                archive_set_error(
                        &a->archive, ARCHIVE_ERRNO_MISC,
@@ -272,8 +273,7 @@ start_over:
        }
        cntlen = _warc_rdlen(buf, eoh - buf);
        if (cntlen < 0) {
-               /* nightmare!  the specs say content-length is mandatory
-                * so I don't feel overly bad stopping the reader here */
+               /* This reader requires Content-Length before processing a record. */
                archive_set_error(
                        &a->archive, EINVAL,
                        "Bad content length");
@@ -281,46 +281,44 @@ start_over:
        }
        rtime = _warc_rdrtm(buf, eoh - buf);
        if (rtime == (time_t)-1) {
-               /* record time is mandatory as per WARC/1.0,
-                * so just barf here, fast and loud */
+               /* This reader requires WARC-Date before processing a record. */
                archive_set_error(
                        &a->archive, EINVAL,
                        "Bad record time");
                return (ARCHIVE_FATAL);
        }
 
-       /* let the world know we're a WARC archive */
+       /* Report this archive as WARC. */
        a->archive.archive_format = ARCHIVE_FORMAT_WARC;
        if (ver != w->pver) {
-               /* stringify this entry's version */
+               /* Format this entry's WARC version. */
                archive_string_sprintf(&w->sver,
                        "WARC/%u.%u", ver / 10000, (ver % 10000) / 100);
-               /* remember the version */
+               /* Remember the version for later entries. */
                w->pver = ver;
        }
-       /* start off with the type */
+       /* Parse the record type. */
        ftyp = _warc_rdtyp(buf, eoh - buf);
-       /* and let future calls know about the content */
+       /* Save content state for subsequent read calls. */
        w->cntlen = cntlen;
        w->cntoff = 0;
-       mtime = 0;/* Avoid compiling error on some platform. */
+       mtime = 0;/* Avoid compiler warnings on some platforms. */
 
        switch (ftyp) {
        case WT_RSRC:
        case WT_RSP:
-               /* only try and read the filename in the cases that are
-                * guaranteed to have one */
+               /* Read the filename only for record types that are expected to
+                * have a target URI. */
                fnam = _warc_rduri(buf, eoh - buf);
-               /* check the last character in the URI to avoid creating
-                * directory endpoints as files, see Todo above */
+               /* Avoid creating directory endpoints as files. */
                if (fnam.len == 0 || fnam.str[fnam.len - 1] == '/') {
-                       /* break here for now */
+                       /* Skip this record. */
                        fnam.len = 0U;
                        fnam.str = NULL;
                        break;
                }
-               /* bang to our string pool, so we save a
-                * malloc()+free() roundtrip */
+               /* Copy the name into the reusable string pool to avoid a malloc/free
+                * roundtrip for each entry. */
                if (fnam.len + 1U > w->pool.len) {
                        w->pool.len = ((fnam.len + 64U) / 64U) * 64U;
                        tmp = realloc(w->pool.str, w->pool.len);
@@ -334,13 +332,11 @@ start_over:
                }
                memcpy(w->pool.str, fnam.str, fnam.len);
                w->pool.str[fnam.len] = '\0';
-               /* let no one else know about the pool, it's a secret, shhh */
+               /* Hide the pool implementation behind the parsed string. */
                fnam.str = w->pool.str;
 
-               /* snarf mtime or deduce from rtime
-                * this is a custom header added by our writer, it's quite
-                * hard to believe anyone else would go through with it
-                * (apart from being part of some http responses of course) */
+               /* Use a Last-Modified record header when present; otherwise fall back
+                * to WARC-Date. */
                if ((mtime = _warc_rdmtm(buf, eoh - buf)) == (time_t)-1) {
                        mtime = rtime;
                }
@@ -359,19 +355,19 @@ start_over:
                break;
        }
 
-       /* now eat some of those delicious buffer bits */
+       /* Consume the record header. */
        __archive_read_consume(a, eoh - buf);
 
        switch (ftyp) {
        case WT_RSRC:
        case WT_RSP:
                if (fnam.len > 0U) {
-                       /* populate entry object */
+                       /* Populate the entry object. */
                        archive_entry_set_filetype(entry, AE_IFREG);
                        archive_entry_copy_pathname(entry, fnam.str);
                        archive_entry_set_size(entry, cntlen);
                        archive_entry_set_perm(entry, 0644);
-                       /* rtime is the new ctime, mtime stays mtime */
+                       /* WARC-Date becomes ctime; mtime comes from Last-Modified or WARC-Date. */
                        archive_entry_set_ctime(entry, rtime, 0L);
                        archive_entry_set_mtime(entry, mtime, 0L);
                        break;
@@ -386,7 +382,7 @@ start_over:
        case WT_CONT:
        case LAST_WT:
        default:
-               /* consume the content and start over */
+               /* Skip this record body and look for the next one. */
                if (_warc_skip(a) < 0)
                        return (ARCHIVE_FATAL);
                goto start_over;
@@ -408,7 +404,7 @@ _warc_read(struct archive_read *a, const void **buf, size_t *bsz, int64_t *off)
 
        if (w->cntoff >= w->cntlen) {
        eof:
-               /* it's our lucky day, no work, we can leave early */
+               /* No data is available to return for this entry. */
                *buf = NULL;
                *bsz = 0U;
                *off = w->cntoff;
@@ -418,12 +414,12 @@ _warc_read(struct archive_read *a, const void **buf, size_t *bsz, int64_t *off)
        rab = __archive_read_ahead(a, 1U, &nrd);
        if (nrd < 0) {
                *bsz = 0U;
-               /* big catastrophe */
+               /* Propagate the read error. */
                return (int)nrd;
        } else if (nrd == 0) {
                goto eof;
        } else if ((int64_t)nrd > w->cntlen - w->cntoff) {
-               /* clamp to content-length */
+               /* Clamp reads to Content-Length. */
                nrd = w->cntlen - w->cntoff;
        }
        *off = w->cntoff;
@@ -455,7 +451,7 @@ _warc_skip(struct archive_read *a)
 }
 
 \f
-/* private routines */
+/* Private routines */
 static void*
 deconst(const void *c)
 {
@@ -475,44 +471,41 @@ xmemmem(const char *hay, const size_t haysize,
        unsigned int nsum;
        unsigned int eqp;
 
-       /* trivial checks first
-         * a 0-sized needle is defined to be found anywhere in haystack
-         * then run strchr() to find a candidate in HAYSTACK (i.e. a portion
-         * that happens to begin with *NEEDLE) */
+       /* Handle trivial cases first.  A zero-sized needle is defined to be
+        * found anywhere in the haystack; otherwise find the first candidate
+        * that begins with *NEEDLE. */
        if (needlesize == 0UL) {
                return deconst(hay);
        } else if ((hay = memchr(hay, *needle, haysize)) == NULL) {
-               /* trivial */
+               /* No candidate match remains. */
                return NULL;
        }
 
-       /* First characters of haystack and needle are the same now. Both are
-        * guaranteed to be at least one character long.  Now computes the sum
-        * of characters values of needle together with the sum of the first
-        * needle_len characters of haystack. */
+       /* The first characters of haystack and needle already match, and both
+        * strings are at least one character long.  Compute the rolling XOR
+        * values for the needle and the first NEEDLESIZE characters of haystack. */
        for (hp = hay + 1U, np = needle + 1U, hsum = *hay, nsum = *hay, eqp = 1U;
             hp < eoh && np < eon;
             hsum ^= *hp, nsum ^= *np, eqp &= *hp == *np, hp++, np++);
 
        /* HP now references the (NEEDLESIZE + 1)-th character. */
        if (np < eon) {
-               /* haystack is smaller than needle, :O */
+               /* The haystack is smaller than the needle. */
                return NULL;
        } else if (eqp) {
-               /* found a match */
+               /* Found a match. */
                return deconst(hay);
        }
 
-       /* now loop through the rest of haystack,
-        * updating the sum iteratively */
+       /* Loop through the rest of the haystack and update the rolling XOR
+        * iteratively. */
        for (cand = hay; hp < eoh; hp++) {
                hsum ^= *cand++;
                hsum ^= *hp;
 
-               /* Since the sum of the characters is already known to be
-                * equal at that point, it is enough to check just NEEDLESIZE - 1
-                * characters for equality,
-                * also CAND is by design < HP, so no need for range checks */
+               /* When the rolling XOR values match, it is enough to check
+                * NEEDLESIZE - 1 characters for equality.  CAND is always before
+                * HP by design, so no range check is needed. */
                if (hsum == nsum && memcmp(cand, needle, needlesize - 1U) == 0) {
                        return deconst(cand);
                }
@@ -525,7 +518,7 @@ strtoi_lim(const char *str, const char **ep, int llim, int ulim)
 {
        int res = 0;
        const char *sp;
-       /* we keep track of the number of digits via rulim */
+       /* Track the number of digits with rulim. */
        int rulim;
 
        for (sp = str, rulim = ulim > 10 ? ulim : 10;
@@ -552,11 +545,11 @@ time_from_tm(struct tm *t)
         /* Use platform timegm() if available. */
         return (timegm(t));
 #else
-        /* Else use direct calculation using POSIX assumptions. */
-        /* First, fix up tm_yday based on the year/month/day. */
+        /* Otherwise, calculate directly using POSIX assumptions. */
+        /* First, fix up tm_yday based on the year, month, and day. */
         if (mktime(t) == (time_t)-1)
                 return ((time_t)-1);
-        /* Then we can compute timegm() from first principles. */
+        /* Then compute timegm() from first principles. */
         return (t->tm_sec
             + t->tm_min * 60
             + t->tm_hour * 3600
@@ -571,48 +564,48 @@ time_from_tm(struct tm *t)
 static time_t
 xstrpisotime(const char *s, char **endptr)
 {
-/** like strptime() but strictly for ISO 8601 Zulu strings */
+/* Like strptime(), but only for ISO 8601 Zulu strings. */
        struct tm tm;
        time_t res = (time_t)-1;
 
-       /* make sure tm is clean */
+       /* Clear the tm structure. */
        memset(&tm, 0, sizeof(tm));
 
-       /* as a courtesy to our callers, and since this is a non-standard
-        * routine, we skip leading whitespace */
+       /* This is a non-standard routine, so skip leading whitespace for
+        * caller convenience. */
        while (*s == ' ' || *s == '\t')
                ++s;
 
-       /* read year */
+       /* Read the year. */
        if ((tm.tm_year = strtoi_lim(s, &s, 1583, 4095)) < 0 || *s++ != '-') {
                goto out;
        }
-       /* read month */
+       /* Read the month. */
        if ((tm.tm_mon = strtoi_lim(s, &s, 1, 12)) < 0 || *s++ != '-') {
                goto out;
        }
-       /* read day-of-month */
+       /* Read the day of the month. */
        if ((tm.tm_mday = strtoi_lim(s, &s, 1, 31)) < 0 || *s++ != 'T') {
                goto out;
        }
-       /* read hour */
+       /* Read the hour. */
        if ((tm.tm_hour = strtoi_lim(s, &s, 0, 23)) < 0 || *s++ != ':') {
                goto out;
        }
-       /* read minute */
+       /* Read the minute. */
        if ((tm.tm_min = strtoi_lim(s, &s, 0, 59)) < 0 || *s++ != ':') {
                goto out;
        }
-       /* read second */
+       /* Read the second. */
        if ((tm.tm_sec = strtoi_lim(s, &s, 0, 60)) < 0 || *s++ != 'Z') {
                goto out;
        }
 
-       /* massage TM to fulfill some of POSIX' constraints */
+       /* Adjust tm fields to satisfy POSIX constraints. */
        tm.tm_year -= 1900;
        tm.tm_mon--;
 
-       /* now convert our custom tm struct to a unix stamp using UTC */
+       /* Convert the tm structure to a Unix timestamp in UTC. */
        res = time_from_tm(&tm);
 
 out:
@@ -631,35 +624,35 @@ _warc_rdver(const char *buf, size_t bsz)
        unsigned int end = 0U;
 
        if (bsz < 12 || memcmp(buf, magic, sizeof(magic) - 1U) != 0) {
-               /* buffer too small or invalid magic */
+               /* Buffer too small or invalid magic. */
                return ver;
        }
-       /* looks good so far, read the version number for a laugh */
+       /* Parse the version number. */
        buf += sizeof(magic) - 1U;
 
        if (isdigit((unsigned char)buf[0U]) && (buf[1U] == '.') &&
            isdigit((unsigned char)buf[2U])) {
-               /* we support a maximum of 2 digits in the minor version */
+               /* Support at most two digits in the minor version. */
                if (isdigit((unsigned char)buf[3U]))
                        end = 1U;
-               /* set up major version */
+               /* Set up the major version. */
                ver = (buf[0U] - '0') * 10000U;
-               /* set up minor version */
+               /* Set up the minor version. */
                if (end == 1U) {
                        ver += (buf[2U] - '0') * 1000U;
                        ver += (buf[3U] - '0') * 100U;
                } else
                        ver += (buf[2U] - '0') * 100U;
                /*
-                * WARC below version 0.12 has a space-separated header
-                * WARC 0.12 and above terminates the version with a CRLF
+                * WARC versions before 0.12 use a space-separated header.
+                * WARC 0.12 and later terminate the version with CRLF.
                 */
                c = buf + 3U + end;
                if (ver >= 1200U) {
                        if (memcmp(c, "\r\n", 2U) != 0)
                                ver = 0U;
                } else {
-                       /* ver < 1200U */
+                       /* Version is below WARC 0.12. */
                        if (*c != ' ' && *c != '\t')
                                ver = 0U;
                }
@@ -674,16 +667,16 @@ _warc_rdtyp(const char *buf, size_t bsz)
        const char *val, *eol;
 
        if ((val = xmemmem(buf, bsz, _key, sizeof(_key) - 1U)) == NULL) {
-               /* no bother */
+               /* Header field is absent. */
                return WT_NONE;
        }
        val += sizeof(_key) - 1U;
        if ((eol = _warc_find_eol(val, buf + bsz - val)) == NULL) {
-               /* no end of line */
+               /* Header field has no end of line. */
                return WT_NONE;
        }
 
-       /* overread whitespace */
+       /* Skip leading whitespace. */
        while (val < eol && (*val == ' ' || *val == '\t'))
                ++val;
 
@@ -704,48 +697,48 @@ _warc_rduri(const char *buf, size_t bsz)
        warc_string_t res = {0U, NULL};
 
        if ((val = xmemmem(buf, bsz, _key, sizeof(_key) - 1U)) == NULL) {
-               /* no bother */
+               /* Header field is absent. */
                return res;
        }
-       /* overread whitespace */
+       /* Skip leading whitespace. */
        val += sizeof(_key) - 1U;
        if ((eol = _warc_find_eol(val, buf + bsz - val)) == NULL) {
-               /* no end of line */
+               /* Header field has no end of line. */
                return res;
        }
 
        while (val < eol && (*val == ' ' || *val == '\t'))
                ++val;
 
-       /* overread URL designators */
+       /* Locate the :// separator. */
        if ((uri = xmemmem(val, eol - val, "://", 3U)) == NULL) {
-               /* not touching that! */
+               /* Ignore values without a :// separator. */
                return res;
        }
 
-       /* spaces inside uri are not allowed, CRLF should follow */
+       /* Spaces inside a URI are not allowed; CRLF should follow. */
        for (p = val; p < eol; p++) {
                if (isspace((unsigned char)*p))
                        return res;
        }
 
-       /* there must be at least space for ftp */
+       /* Require enough room for the shortest supported scheme. */
        if (uri < (val + 3U))
                return res;
 
-       /* move uri to point to after :// */
+       /* Move uri past the :// separator. */
        uri += 3U;
 
-       /* now then, inspect the URI */
+       /* Inspect the scheme prefix. */
        if (memcmp(val, "file", 4U) == 0) {
-               /* perfect, nothing left to do here */
+               /* Keep file:// paths as-is. */
 
        } else if (memcmp(val, "http", 4U) == 0 ||
                   memcmp(val, "ftp", 3U) == 0) {
-               /* overread domain, and the first / */
+               /* Skip the domain and the first slash. */
                while (uri < eol && *uri++ != '/');
        } else {
-               /* not sure what to do? best to bugger off */
+               /* Unsupported URI scheme. */
                return res;
        }
        res.str = uri;
@@ -761,21 +754,21 @@ _warc_rdlen(const char *buf, size_t bsz)
        int64_t len;
 
        if ((val = xmemmem(buf, bsz, _key, sizeof(_key) - 1U)) == NULL) {
-               /* no bother */
+               /* Header field is absent. */
                return -1;
        }
        val += sizeof(_key) - 1U;
 
        if ((eol = _warc_find_eol(val, buf + bsz - val)) == NULL) {
-               /* malformed, no end-of-line */
+               /* Malformed field with no end of line. */
                return -1;
        }
 
-       /* skip leading whitespace */
+       /* Skip leading whitespace. */
        while (val < eol && (*val == ' ' || *val == '\t'))
                val++;
 
-       /* there must be at least one digit */
+       /* Require at least one digit. */
        if (val >= eol || *val < '0' || *val > '9')
                return -1;
 
@@ -803,19 +796,19 @@ _warc_rdrtm(const char *buf, size_t bsz)
        time_t res;
 
        if ((val = xmemmem(buf, bsz, _key, sizeof(_key) - 1U)) == NULL) {
-               /* no bother */
+               /* Header field is absent. */
                return (time_t)-1;
        }
        val += sizeof(_key) - 1U;
        if ((eol = _warc_find_eol(val, buf + bsz - val)) == NULL ) {
-               /* no end of line */
+               /* Header field has no end of line. */
                return -1;
        }
 
-       /* xstrpisotime() kindly overreads whitespace for us, so use that */
+       /* xstrpisotime() skips leading whitespace. */
        res = xstrpisotime(val, &on);
        if (on != eol) {
-               /* line must end here */
+               /* The field must end here. */
                return -1;
        }
        return res;
@@ -830,19 +823,19 @@ _warc_rdmtm(const char *buf, size_t bsz)
        time_t res;
 
        if ((val = xmemmem(buf, bsz, _key, sizeof(_key) - 1U)) == NULL) {
-               /* no bother */
+               /* Header field is absent. */
                return (time_t)-1;
        }
        val += sizeof(_key) - 1U;
        if ((eol = _warc_find_eol(val, buf + bsz - val)) == NULL ) {
-               /* no end of line */
+               /* Header field has no end of line. */
                return -1;
        }
 
-       /* xstrpisotime() kindly overreads whitespace for us, so use that */
+       /* xstrpisotime() skips leading whitespace. */
        res = xstrpisotime(val, &on);
        if (on != eol) {
-               /* line must end here */
+               /* The field must end here. */
                return -1;
        }
        return res;
@@ -868,4 +861,3 @@ _warc_find_eol(const char *buf, size_t bsz)
 
        return hit;
 }
-/* archive_read_support_format_warc.c ends here */