[thirdparty/git.git] / pretty.c

#include "cache.h"
#include "commit.h"
#include "utf8.h"
#include "diff.h"
#include "revision.h"

static struct cmt_fmt_map {
	const char *n;
	size_t cmp_len;
	enum cmit_fmt v;
} cmt_fmts[] = {
	{ "raw",	1,	CMIT_FMT_RAW },
	{ "medium",	1,	CMIT_FMT_MEDIUM },
	{ "short",	1,	CMIT_FMT_SHORT },
	{ "email",	1,	CMIT_FMT_EMAIL },
	{ "full",	5,	CMIT_FMT_FULL },
	{ "fuller",	5,	CMIT_FMT_FULLER },
	{ "oneline",	1,	CMIT_FMT_ONELINE },
	{ "format:",	7,	CMIT_FMT_USERFORMAT},
};

static char *user_format;

enum cmit_fmt get_commit_format(const char *arg)
{
	int i;

	if (!arg || !*arg)
		return CMIT_FMT_DEFAULT;
	if (*arg == '=')
		arg++;
	if (!prefixcmp(arg, "format:")) {
		if (user_format)
			free(user_format);
		user_format = xstrdup(arg + 7);
		return CMIT_FMT_USERFORMAT;
	}
	for (i = 0; i < ARRAY_SIZE(cmt_fmts); i++) {
		if (!strncmp(arg, cmt_fmts[i].n, cmt_fmts[i].cmp_len) &&
		    !strncmp(arg, cmt_fmts[i].n, strlen(arg)))
			return cmt_fmts[i].v;
	}

	die("invalid --pretty format: %s", arg);
}

/*
 * Generic support for pretty-printing the header
 */
static int get_one_line(const char *msg)
{
	int ret = 0;

	for (;;) {
		char c = *msg++;
		if (!c)
			break;
		ret++;
		if (c == '\n')
			break;
	}
	return ret;
}

/* High bit set, or ISO-2022-INT */
int non_ascii(int ch)
{
	ch = (ch & 0xff);
	return ((ch & 0x80) || (ch == 0x1b));
}

static int is_rfc2047_special(char ch)
{
	return (non_ascii(ch) || (ch == '=') || (ch == '?') || (ch == '_'));
}

static void add_rfc2047(struct strbuf *sb, const char *line, int len,
		       const char *encoding)
{
	int i, last;

	for (i = 0; i < len; i++) {
		int ch = line[i];
		if (non_ascii(ch))
			goto needquote;
		if ((i + 1 < len) && (ch == '=' && line[i+1] == '?'))
			goto needquote;
	}
	strbuf_add(sb, line, len);
	return;

needquote:
	strbuf_grow(sb, len * 3 + strlen(encoding) + 100);
	strbuf_addf(sb, "=?%s?q?", encoding);
	for (i = last = 0; i < len; i++) {
		unsigned ch = line[i] & 0xFF;
		/*
		 * We encode ' ' using '=20' even though rfc2047
		 * allows using '_' for readability.  Unfortunately,
		 * many programs do not understand this and just
		 * leave the underscore in place.
		 */
		if (is_rfc2047_special(ch) || ch == ' ') {
			strbuf_add(sb, line + last, i - last);
			strbuf_addf(sb, "=%02X", ch);
			last = i + 1;
		}
	}
	strbuf_add(sb, line + last, len - last);
	strbuf_addstr(sb, "?=");
}

static void add_user_info(const char *what, enum cmit_fmt fmt, struct strbuf *sb,
			 const char *line, enum date_mode dmode,
			 const char *encoding)
{
	char *date;
	int namelen;
	unsigned long time;
	int tz;
	const char *filler = "    ";

	if (fmt == CMIT_FMT_ONELINE)
		return;
	date = strchr(line, '>');
	if (!date)
		return;
	namelen = ++date - line;
	time = strtoul(date, &date, 10);
	tz = strtol(date, NULL, 10);

	if (fmt == CMIT_FMT_EMAIL) {
		char *name_tail = strchr(line, '<');
		int display_name_length;
		if (!name_tail)
			return;
		while (line < name_tail && isspace(name_tail[-1]))
			name_tail--;
		display_name_length = name_tail - line;
		filler = "";
		strbuf_addstr(sb, "From: ");
		add_rfc2047(sb, line, display_name_length, encoding);
		strbuf_add(sb, name_tail, namelen - display_name_length);
		strbuf_addch(sb, '\n');
	} else {
		strbuf_addf(sb, "%s: %.*s%.*s\n", what,
			      (fmt == CMIT_FMT_FULLER) ? 4 : 0,
			      filler, namelen, line);
	}
	switch (fmt) {
	case CMIT_FMT_MEDIUM:
		strbuf_addf(sb, "Date:   %s\n", show_date(time, tz, dmode));
		break;
	case CMIT_FMT_EMAIL:
		strbuf_addf(sb, "Date: %s\n", show_date(time, tz, DATE_RFC2822));
		break;
	case CMIT_FMT_FULLER:
		strbuf_addf(sb, "%sDate: %s\n", what, show_date(time, tz, dmode));
		break;
	default:
		/* notin' */
		break;
	}
}

static int is_empty_line(const char *line, int *len_p)
{
	int len = *len_p;
	while (len && isspace(line[len-1]))
		len--;
	*len_p = len;
	return !len;
}

static void add_merge_info(enum cmit_fmt fmt, struct strbuf *sb,
			const struct commit *commit, int abbrev)
{
	struct commit_list *parent = commit->parents;

	if ((fmt == CMIT_FMT_ONELINE) || (fmt == CMIT_FMT_EMAIL) ||
	    !parent || !parent->next)
		return;

	strbuf_addstr(sb, "Merge:");

	while (parent) {
		struct commit *p = parent->item;
		const char *hex = NULL;
		const char *dots;
		if (abbrev)
			hex = find_unique_abbrev(p->object.sha1, abbrev);
		if (!hex)
			hex = sha1_to_hex(p->object.sha1);
		dots = (abbrev && strlen(hex) != 40) ?  "..." : "";
		parent = parent->next;

		strbuf_addf(sb, " %s%s", hex, dots);
	}
	strbuf_addch(sb, '\n');
}

static char *get_header(const struct commit *commit, const char *key)
{
	int key_len = strlen(key);
	const char *line = commit->buffer;

	for (;;) {
		const char *eol = strchr(line, '\n'), *next;

		if (line == eol)
			return NULL;
		if (!eol) {
			eol = line + strlen(line);
			next = NULL;
		} else
			next = eol + 1;
		if (eol - line > key_len &&
		    !strncmp(line, key, key_len) &&
		    line[key_len] == ' ') {
			return xmemdupz(line + key_len + 1, eol - line - key_len - 1);
		}
		line = next;
	}
}

static char *replace_encoding_header(char *buf, const char *encoding)
{
	struct strbuf tmp;
	size_t start, len;
	char *cp = buf;

	/* guess if there is an encoding header before a \n\n */
	while (strncmp(cp, "encoding ", strlen("encoding "))) {
		cp = strchr(cp, '\n');
		if (!cp || *++cp == '\n')
			return buf;
	}
	start = cp - buf;
	cp = strchr(cp, '\n');
	if (!cp)
		return buf; /* should not happen but be defensive */
	len = cp + 1 - (buf + start);

	strbuf_init(&tmp, 0);
	strbuf_attach(&tmp, buf, strlen(buf), strlen(buf) + 1);
	if (is_encoding_utf8(encoding)) {
		/* we have re-coded to UTF-8; drop the header */
		strbuf_remove(&tmp, start, len);
	} else {
		/* just replaces XXXX in 'encoding XXXX\n' */
		strbuf_splice(&tmp, start + strlen("encoding "),
					  len - strlen("encoding \n"),
					  encoding, strlen(encoding));
	}
	return strbuf_detach(&tmp, NULL);
}

static char *logmsg_reencode(const struct commit *commit,
			     const char *output_encoding)
{
	static const char *utf8 = "utf-8";
	const char *use_encoding;
	char *encoding;
	char *out;

	if (!*output_encoding)
		return NULL;
	encoding = get_header(commit, "encoding");
	use_encoding = encoding ? encoding : utf8;
	if (!strcmp(use_encoding, output_encoding))
		if (encoding) /* we'll strip encoding header later */
			out = xstrdup(commit->buffer);
		else
			return NULL; /* nothing to do */
	else
		out = reencode_string(commit->buffer,
				      output_encoding, use_encoding);
	if (out)
		out = replace_encoding_header(out, output_encoding);

	free(encoding);
	return out;
}

static void format_person_part(struct strbuf *sb, char part,
                               const char *msg, int len)
{
	int start, end, tz = 0;
	unsigned long date;
	char *ep;

	/* parse name */
	for (end = 0; end < len && msg[end] != '<'; end++)
		; /* do nothing */
	start = end + 1;
	while (end > 0 && isspace(msg[end - 1]))
		end--;
	if (part == 'n') {	/* name */
		strbuf_add(sb, msg, end);
		return;
	}

	if (start >= len)
		return;

	/* parse email */
	for (end = start + 1; end < len && msg[end] != '>'; end++)
		; /* do nothing */

	if (end >= len)
		return;

	if (part == 'e') {	/* email */
		strbuf_add(sb, msg + start, end - start);
		return;
	}

	/* parse date */
	for (start = end + 1; start < len && isspace(msg[start]); start++)
		; /* do nothing */
	if (start >= len)
		return;
	date = strtoul(msg + start, &ep, 10);
	if (msg + start == ep)
		return;

	if (part == 't') {	/* date, UNIX timestamp */
		strbuf_add(sb, msg + start, ep - (msg + start));
		return;
	}

	/* parse tz */
	for (start = ep - msg + 1; start < len && isspace(msg[start]); start++)
		; /* do nothing */
	if (start + 1 < len) {
		tz = strtoul(msg + start + 1, NULL, 10);
		if (msg[start] == '-')
			tz = -tz;
	}

	switch (part) {
	case 'd':	/* date */
		strbuf_addstr(sb, show_date(date, tz, DATE_NORMAL));
		return;
	case 'D':	/* date, RFC2822 style */
		strbuf_addstr(sb, show_date(date, tz, DATE_RFC2822));
		return;
	case 'r':	/* date, relative */
		strbuf_addstr(sb, show_date(date, tz, DATE_RELATIVE));
		return;
	case 'i':	/* date, ISO 8601 */
		strbuf_addstr(sb, show_date(date, tz, DATE_ISO8601));
		return;
	}
}

static void format_commit_item(struct strbuf *sb, const char *placeholder,
                               void *context)
{
	const struct commit *commit = context;
	struct commit_list *p;
	int i;
	enum { HEADER, SUBJECT, BODY } state;
	const char *msg = commit->buffer;

	/* these are independent of the commit */
	switch (placeholder[0]) {
	case 'C':
		switch (placeholder[3]) {
		case 'd':	/* red */
			strbuf_addstr(sb, "\033[31m");
			return;
		case 'e':	/* green */
			strbuf_addstr(sb, "\033[32m");
			return;
		case 'u':	/* blue */
			strbuf_addstr(sb, "\033[34m");
			return;
		case 's':	/* reset color */
			strbuf_addstr(sb, "\033[m");
			return;
		}
	case 'n':		/* newline */
		strbuf_addch(sb, '\n');
		return;
	}

	/* these depend on the commit */
	if (!commit->object.parsed)
		parse_object(commit->object.sha1);

	switch (placeholder[0]) {
	case 'H':		/* commit hash */
		strbuf_addstr(sb, sha1_to_hex(commit->object.sha1));
		return;
	case 'h':		/* abbreviated commit hash */
		strbuf_addstr(sb, find_unique_abbrev(commit->object.sha1,
		                                     DEFAULT_ABBREV));
		return;
	case 'T':		/* tree hash */
		strbuf_addstr(sb, sha1_to_hex(commit->tree->object.sha1));
		return;
	case 't':		/* abbreviated tree hash */
		strbuf_addstr(sb, find_unique_abbrev(commit->tree->object.sha1,
		                                     DEFAULT_ABBREV));
		return;
	case 'P':		/* parent hashes */
		for (p = commit->parents; p; p = p->next) {
			if (p != commit->parents)
				strbuf_addch(sb, ' ');
			strbuf_addstr(sb, sha1_to_hex(p->item->object.sha1));
		}
		return;
	case 'p':		/* abbreviated parent hashes */
		for (p = commit->parents; p; p = p->next) {
			if (p != commit->parents)
				strbuf_addch(sb, ' ');
			strbuf_addstr(sb, find_unique_abbrev(
					p->item->object.sha1, DEFAULT_ABBREV));
		}
		return;
	case 'm':		/* left/right/bottom */
		strbuf_addch(sb, (commit->object.flags & BOUNDARY)
		                 ? '-'
		                 : (commit->object.flags & SYMMETRIC_LEFT)
		                 ? '<'
		                 : '>');
		return;
	}

	/* For the rest we have to parse the commit header. */
	for (i = 0, state = HEADER; msg[i] && state < BODY; i++) {
		int eol;
		for (eol = i; msg[eol] && msg[eol] != '\n'; eol++)
			; /* do nothing */

		if (state == SUBJECT) {
			if (placeholder[0] == 's') {
				strbuf_add(sb, msg + i, eol - i);
				return;
			}
			i = eol;
		}
		if (i == eol) {
			state++;
			/* strip empty lines */
			while (msg[eol + 1] == '\n')
				eol++;
		} else if (!prefixcmp(msg + i, "author ")) {
			if (placeholder[0] == 'a') {
				format_person_part(sb, placeholder[1],
				                   msg + i + 7, eol - i - 7);
				return;
			}
		} else if (!prefixcmp(msg + i, "committer ")) {
			if (placeholder[0] == 'c') {
				format_person_part(sb, placeholder[1],
				                   msg + i + 10, eol - i - 10);
				return;
			}
		} else if (!prefixcmp(msg + i, "encoding ")) {
			if (placeholder[0] == 'e') {
				strbuf_add(sb, msg + i + 9, eol - i - 9);
				return;
			}
		}
		i = eol;
	}
	if (msg[i] && placeholder[0] == 'b')	/* body */
		strbuf_addstr(sb, msg + i);
}

void format_commit_message(const struct commit *commit,
                           const void *format, struct strbuf *sb)
{
	const char *placeholders[] = {
		"H",		/* commit hash */
		"h",		/* abbreviated commit hash */
		"T",		/* tree hash */
		"t",		/* abbreviated tree hash */
		"P",		/* parent hashes */
		"p",		/* abbreviated parent hashes */
		"an",		/* author name */
		"ae",		/* author email */
		"ad",		/* author date */
		"aD",		/* author date, RFC2822 style */
		"ar",		/* author date, relative */
		"at",		/* author date, UNIX timestamp */
		"ai",		/* author date, ISO 8601 */
		"cn",		/* committer name */
		"ce",		/* committer email */
		"cd",		/* committer date */
		"cD",		/* committer date, RFC2822 style */
		"cr",		/* committer date, relative */
		"ct",		/* committer date, UNIX timestamp */
		"ci",		/* committer date, ISO 8601 */
		"e",		/* encoding */
		"s",		/* subject */
		"b",		/* body */
		"Cred",		/* red */
		"Cgreen",	/* green */
		"Cblue",	/* blue */
		"Creset",	/* reset color */
		"n",		/* newline */
		"m",		/* left/right/bottom */
		NULL
	};
	strbuf_expand(sb, format, placeholders, format_commit_item, (void *)commit);
}

static void pp_header(enum cmit_fmt fmt,
		      int abbrev,
		      enum date_mode dmode,
		      const char *encoding,
		      const struct commit *commit,
		      const char **msg_p,
		      struct strbuf *sb)
{
	int parents_shown = 0;

	for (;;) {
		const char *line = *msg_p;
		int linelen = get_one_line(*msg_p);

		if (!linelen)
			return;
		*msg_p += linelen;

		if (linelen == 1)
			/* End of header */
			return;

		if (fmt == CMIT_FMT_RAW) {
			strbuf_add(sb, line, linelen);
			continue;
		}

		if (!memcmp(line, "parent ", 7)) {
			if (linelen != 48)
				die("bad parent line in commit");
			continue;
		}

		if (!parents_shown) {
			struct commit_list *parent;
			int num;
			for (parent = commit->parents, num = 0;
			     parent;
			     parent = parent->next, num++)
				;
			/* with enough slop */
			strbuf_grow(sb, num * 50 + 20);
			add_merge_info(fmt, sb, commit, abbrev);
			parents_shown = 1;
		}

		/*
		 * MEDIUM == DEFAULT shows only author with dates.
		 * FULL shows both authors but not dates.
		 * FULLER shows both authors and dates.
		 */
		if (!memcmp(line, "author ", 7)) {
			strbuf_grow(sb, linelen + 80);
			add_user_info("Author", fmt, sb, line + 7, dmode, encoding);
		}
		if (!memcmp(line, "committer ", 10) &&
		    (fmt == CMIT_FMT_FULL || fmt == CMIT_FMT_FULLER)) {
			strbuf_grow(sb, linelen + 80);
			add_user_info("Commit", fmt, sb, line + 10, dmode, encoding);
		}
	}
}

static void pp_title_line(enum cmit_fmt fmt,
			  const char **msg_p,
			  struct strbuf *sb,
			  const char *subject,
			  const char *after_subject,
			  const char *encoding,
			  int plain_non_ascii)
{
	struct strbuf title;

	strbuf_init(&title, 80);

	for (;;) {
		const char *line = *msg_p;
		int linelen = get_one_line(line);

		*msg_p += linelen;
		if (!linelen || is_empty_line(line, &linelen))
			break;

		strbuf_grow(&title, linelen + 2);
		if (title.len) {
			if (fmt == CMIT_FMT_EMAIL) {
				strbuf_addch(&title, '\n');
			}
			strbuf_addch(&title, ' ');
		}
		strbuf_add(&title, line, linelen);
	}

	strbuf_grow(sb, title.len + 1024);
	if (subject) {
		strbuf_addstr(sb, subject);
		add_rfc2047(sb, title.buf, title.len, encoding);
	} else {
		strbuf_addbuf(sb, &title);
	}
	strbuf_addch(sb, '\n');

	if (plain_non_ascii) {
		const char *header_fmt =
			"MIME-Version: 1.0\n"
			"Content-Type: text/plain; charset=%s\n"
			"Content-Transfer-Encoding: 8bit\n";
		strbuf_addf(sb, header_fmt, encoding);
	}
	if (after_subject) {
		strbuf_addstr(sb, after_subject);
	}
	if (fmt == CMIT_FMT_EMAIL) {
		strbuf_addch(sb, '\n');
	}
	strbuf_release(&title);
}

static void pp_remainder(enum cmit_fmt fmt,
			 const char **msg_p,
			 struct strbuf *sb,
			 int indent)
{
	int first = 1;
	for (;;) {
		const char *line = *msg_p;
		int linelen = get_one_line(line);
		*msg_p += linelen;

		if (!linelen)
			break;

		if (is_empty_line(line, &linelen)) {
			if (first)
				continue;
			if (fmt == CMIT_FMT_SHORT)
				break;
		}
		first = 0;

		strbuf_grow(sb, linelen + indent + 20);
		if (indent) {
			memset(sb->buf + sb->len, ' ', indent);
			strbuf_setlen(sb, sb->len + indent);
		}
		strbuf_add(sb, line, linelen);
		strbuf_addch(sb, '\n');
	}
}

void pretty_print_commit(enum cmit_fmt fmt, const struct commit *commit,
				  struct strbuf *sb, int abbrev,
				  const char *subject, const char *after_subject,
				  enum date_mode dmode, int plain_non_ascii)
{
	unsigned long beginning_of_body;
	int indent = 4;
	const char *msg = commit->buffer;
	char *reencoded;
	const char *encoding;

	if (fmt == CMIT_FMT_USERFORMAT) {
		format_commit_message(commit, user_format, sb);
		return;
	}

	encoding = (git_log_output_encoding
		    ? git_log_output_encoding
		    : git_commit_encoding);
	if (!encoding)
		encoding = "utf-8";
	reencoded = logmsg_reencode(commit, encoding);
	if (reencoded) {
		msg = reencoded;
	}

	if (fmt == CMIT_FMT_ONELINE || fmt == CMIT_FMT_EMAIL)
		indent = 0;

	/* After-subject is used to pass in Content-Type: multipart
	 * MIME header; in that case we do not have to do the
	 * plaintext content type even if the commit message has
	 * non 7-bit ASCII character.  Otherwise, check if we need
	 * to say this is not a 7-bit ASCII.
	 */
	if (fmt == CMIT_FMT_EMAIL && !after_subject) {
		int i, ch, in_body;

		for (in_body = i = 0; (ch = msg[i]); i++) {
			if (!in_body) {
				/* author could be non 7-bit ASCII but
				 * the log may be so; skip over the
				 * header part first.
				 */
				if (ch == '\n' && msg[i+1] == '\n')
					in_body = 1;
			}
			else if (non_ascii(ch)) {
				plain_non_ascii = 1;
				break;
			}
		}
	}

	pp_header(fmt, abbrev, dmode, encoding, commit, &msg, sb);
	if (fmt != CMIT_FMT_ONELINE && !subject) {
		strbuf_addch(sb, '\n');
	}

	/* Skip excess blank lines at the beginning of body, if any... */
	for (;;) {
		int linelen = get_one_line(msg);
		int ll = linelen;
		if (!linelen)
			break;
		if (!is_empty_line(msg, &ll))
			break;
		msg += linelen;
	}

	/* These formats treat the title line specially. */
	if (fmt == CMIT_FMT_ONELINE || fmt == CMIT_FMT_EMAIL)
		pp_title_line(fmt, &msg, sb, subject,
			      after_subject, encoding, plain_non_ascii);

	beginning_of_body = sb->len;
	if (fmt != CMIT_FMT_ONELINE)
		pp_remainder(fmt, &msg, sb, indent);
	strbuf_rtrim(sb);

	/* Make sure there is an EOLN for the non-oneline case */
	if (fmt != CMIT_FMT_ONELINE)
		strbuf_addch(sb, '\n');

	/*
	 * The caller may append additional body text in e-mail
	 * format.  Make sure we did not strip the blank line
	 * between the header and the body.
	 */
	if (fmt == CMIT_FMT_EMAIL && sb->len <= beginning_of_body)
		strbuf_addch(sb, '\n');
	free(reencoded);
}
Commit	Line	Data
93fc05eb JS	1	#include "cache.h"
93fc05eb JS	2	#include "commit.h"
93fc05eb JS	3	#include "utf8.h"
	4	#include "diff.h"
	5	#include "revision.h"
	6
	7	static struct cmt_fmt_map {
	8	const char *n;
	9	size_t cmp_len;
	10	enum cmit_fmt v;
	11	} cmt_fmts[] = {
	12	{ "raw", 1, CMIT_FMT_RAW },
	13	{ "medium", 1, CMIT_FMT_MEDIUM },
	14	{ "short", 1, CMIT_FMT_SHORT },
	15	{ "email", 1, CMIT_FMT_EMAIL },
	16	{ "full", 5, CMIT_FMT_FULL },
	17	{ "fuller", 5, CMIT_FMT_FULLER },
	18	{ "oneline", 1, CMIT_FMT_ONELINE },
	19	{ "format:", 7, CMIT_FMT_USERFORMAT},
	20	};
	21
	22	static char *user_format;
	23
	24	enum cmit_fmt get_commit_format(const char *arg)
	25	{
	26	int i;
	27
	28	if (!arg \|\| !*arg)
	29	return CMIT_FMT_DEFAULT;
	30	if (*arg == '=')
	31	arg++;
	32	if (!prefixcmp(arg, "format:")) {
	33	if (user_format)
	34	free(user_format);
	35	user_format = xstrdup(arg + 7);
	36	return CMIT_FMT_USERFORMAT;
	37	}
	38	for (i = 0; i < ARRAY_SIZE(cmt_fmts); i++) {
	39	if (!strncmp(arg, cmt_fmts[i].n, cmt_fmts[i].cmp_len) &&
	40	!strncmp(arg, cmt_fmts[i].n, strlen(arg)))
	41	return cmt_fmts[i].v;
	42	}
	43
	44	die("invalid --pretty format: %s", arg);
	45	}
	46
	47	/*
	48	* Generic support for pretty-printing the header
	49	*/
	50	static int get_one_line(const char *msg)
	51	{
	52	int ret = 0;
	53
	54	for (;;) {
	55	char c = *msg++;
	56	if (!c)
	57	break;
	58	ret++;
	59	if (c == '\n')
	60	break;
	61	}
	62	return ret;
	63	}
	64
	65	/* High bit set, or ISO-2022-INT */
	66	int non_ascii(int ch)
67	{
68	ch = (ch & 0xff);
69	return ((ch & 0x80) \|\| (ch == 0x1b));
70	}
71
72	static int is_rfc2047_special(char ch)
73	{
74	return (non_ascii(ch) \|\| (ch == '=') \|\| (ch == '?') \|\| (ch == '_'));
75	}
76
77	static void add_rfc2047(struct strbuf sb, const char line, int len,
78	const char *encoding)
79	{
80	int i, last;
81
82	for (i = 0; i < len; i++) {
83	int ch = line[i];
84	if (non_ascii(ch))
85	goto needquote;
86	if ((i + 1 < len) && (ch == '=' && line[i+1] == '?'))
87	goto needquote;
88	}
89	strbuf_add(sb, line, len);
90	return;
91
92	needquote:
93	strbuf_grow(sb, len * 3 + strlen(encoding) + 100);
94	strbuf_addf(sb, "=?%s?q?", encoding);
95	for (i = last = 0; i < len; i++) {
96	unsigned ch = line[i] & 0xFF;
97	/*
98	* We encode ' ' using '=20' even though rfc2047
99	* allows using '_' for readability. Unfortunately,
100	* many programs do not understand this and just
101	* leave the underscore in place.
102	*/
103	if (is_rfc2047_special(ch) \|\| ch == ' ') {
104	strbuf_add(sb, line + last, i - last);
105	strbuf_addf(sb, "=%02X", ch);
106	last = i + 1;
107	}
108	}
109	strbuf_add(sb, line + last, len - last);
110	strbuf_addstr(sb, "?=");
111	}
112
113	static void add_user_info(const char what, enum cmit_fmt fmt, struct strbuf sb,
114	const char *line, enum date_mode dmode,
115	const char *encoding)
116	{
117	char *date;
118	int namelen;
119	unsigned long time;
120	int tz;
121	const char *filler = " ";
122
123	if (fmt == CMIT_FMT_ONELINE)
124	return;
125	date = strchr(line, '>');
126	if (!date)
127	return;
128	namelen = ++date - line;
129	time = strtoul(date, &date, 10);
130	tz = strtol(date, NULL, 10);
131
132	if (fmt == CMIT_FMT_EMAIL) {
133	char *name_tail = strchr(line, '<');
134	int display_name_length;
135	if (!name_tail)
136	return;
137	while (line < name_tail && isspace(name_tail[-1]))
138	name_tail--;
139	display_name_length = name_tail - line;
140	filler = "";
141	strbuf_addstr(sb, "From: ");
142	add_rfc2047(sb, line, display_name_length, encoding);
143	strbuf_add(sb, name_tail, namelen - display_name_length);
144	strbuf_addch(sb, '\n');
145	} else {
146	strbuf_addf(sb, "%s: %.s%.s\n", what,
147	(fmt == CMIT_FMT_FULLER) ? 4 : 0,
148	filler, namelen, line);
149	}
150	switch (fmt) {
151	case CMIT_FMT_MEDIUM:
152	strbuf_addf(sb, "Date: %s\n", show_date(time, tz, dmode));
153	break;
154	case CMIT_FMT_EMAIL:
155	strbuf_addf(sb, "Date: %s\n", show_date(time, tz, DATE_RFC2822));
156	break;
157	case CMIT_FMT_FULLER:
158	strbuf_addf(sb, "%sDate: %s\n", what, show_date(time, tz, dmode));
159	break;
160	default:
161	/* notin' */
162	break;
163	}
164	}
165
166	static int is_empty_line(const char line, int len_p)
167	{
168	int len = *len_p;
169	while (len && isspace(line[len-1]))
170	len--;
171	*len_p = len;
172	return !len;
173	}
174
175	static void add_merge_info(enum cmit_fmt fmt, struct strbuf *sb,
176	const struct commit *commit, int abbrev)
177	{
178	struct commit_list *parent = commit->parents;
179
180	if ((fmt == CMIT_FMT_ONELINE) \|\| (fmt == CMIT_FMT_EMAIL) \|\|
181	!parent \|\| !parent->next)
182	return;
183
184	strbuf_addstr(sb, "Merge:");
185
186	while (parent) {
187	struct commit *p = parent->item;
188	const char *hex = NULL;
189	const char *dots;
190	if (abbrev)
191	hex = find_unique_abbrev(p->object.sha1, abbrev);
192	if (!hex)
193	hex = sha1_to_hex(p->object.sha1);
194	dots = (abbrev && strlen(hex) != 40) ? "..." : "";
195	parent = parent->next;
196
197	strbuf_addf(sb, " %s%s", hex, dots);
198	}
199	strbuf_addch(sb, '\n');
200	}
201
202	static char get_header(const struct commit commit, const char *key)
203	{
204	int key_len = strlen(key);
205	const char *line = commit->buffer;
206
207	for (;;) {
208	const char eol = strchr(line, '\n'), next;
209
210	if (line == eol)
211	return NULL;
212	if (!eol) {
213	eol = line + strlen(line);
214	next = NULL;
215	} else
216	next = eol + 1;
217	if (eol - line > key_len &&
218	!strncmp(line, key, key_len) &&
219	line[key_len] == ' ') {
220	return xmemdupz(line + key_len + 1, eol - line - key_len - 1);
221	}
222	line = next;
223	}
224	}
225
226	static char replace_encoding_header(char buf, const char *encoding)
227	{
228	struct strbuf tmp;
229	size_t start, len;
230	char *cp = buf;
231
232	/* guess if there is an encoding header before a \n\n */
233	while (strncmp(cp, "encoding ", strlen("encoding "))) {
234	cp = strchr(cp, '\n');
235	if (!cp \|\| *++cp == '\n')
236	return buf;
237	}
238	start = cp - buf;
239	cp = strchr(cp, '\n');
240	if (!cp)
241	return buf; /* should not happen but be defensive */
242	len = cp + 1 - (buf + start);
243
244	strbuf_init(&tmp, 0);
245	strbuf_attach(&tmp, buf, strlen(buf), strlen(buf) + 1);
246	if (is_encoding_utf8(encoding)) {
247	/* we have re-coded to UTF-8; drop the header */
248	strbuf_remove(&tmp, start, len);
249	} else {
250	/* just replaces XXXX in 'encoding XXXX\n' */
251	strbuf_splice(&tmp, start + strlen("encoding "),
252	len - strlen("encoding \n"),
253	encoding, strlen(encoding));
254	}
255	return strbuf_detach(&tmp, NULL);
256	}
257
258	static char logmsg_reencode(const struct commit commit,
259	const char *output_encoding)
260	{
261	static const char *utf8 = "utf-8";
262	const char *use_encoding;
263	char *encoding;
264	char *out;
265
266	if (!*output_encoding)
267	return NULL;
268	encoding = get_header(commit, "encoding");
269	use_encoding = encoding ? encoding : utf8;
270	if (!strcmp(use_encoding, output_encoding))
271	if (encoding) /* we'll strip encoding header later */
272	out = xstrdup(commit->buffer);
273	else
274	return NULL; /* nothing to do */
275	else
276	out = reencode_string(commit->buffer,
277	output_encoding, use_encoding);
278	if (out)
279	out = replace_encoding_header(out, output_encoding);
280
281	free(encoding);
282	return out;
283	}
284
cde75e59 RS	285	static void format_person_part(struct strbuf *sb, char part,
cde75e59 RS	286	const char *msg, int len)
93fc05eb JS	287	{
	288	int start, end, tz = 0;
	289	unsigned long date;
	290	char *ep;
	291
	292	/* parse name */
	293	for (end = 0; end < len && msg[end] != '<'; end++)
	294	; /* do nothing */
	295	start = end + 1;
	296	while (end > 0 && isspace(msg[end - 1]))
	297	end--;
cde75e59 RS	298	if (part == 'n') { /* name */
	299	strbuf_add(sb, msg, end);
	300	return;
	301	}
93fc05eb JS	302
	303	if (start >= len)
	304	return;
	305
	306	/* parse email */
	307	for (end = start + 1; end < len && msg[end] != '>'; end++)
	308	; /* do nothing */
	309
	310	if (end >= len)
	311	return;
	312
cde75e59 RS	313	if (part == 'e') { /* email */
	314	strbuf_add(sb, msg + start, end - start);
	315	return;
	316	}
93fc05eb JS	317
	318	/* parse date */
	319	for (start = end + 1; start < len && isspace(msg[start]); start++)
	320	; /* do nothing */
	321	if (start >= len)
	322	return;
	323	date = strtoul(msg + start, &ep, 10);
	324	if (msg + start == ep)
	325	return;
	326
cde75e59 RS	327	if (part == 't') { /* date, UNIX timestamp */
	328	strbuf_add(sb, msg + start, ep - (msg + start));
	329	return;
	330	}
93fc05eb JS	331
	332	/* parse tz */
	333	for (start = ep - msg + 1; start < len && isspace(msg[start]); start++)
	334	; /* do nothing */
	335	if (start + 1 < len) {
	336	tz = strtoul(msg + start + 1, NULL, 10);
	337	if (msg[start] == '-')
	338	tz = -tz;
	339	}
	340
cde75e59 RS	341	switch (part) {
	342	case 'd': /* date */
	343	strbuf_addstr(sb, show_date(date, tz, DATE_NORMAL));
	344	return;
	345	case 'D': /* date, RFC2822 style */
	346	strbuf_addstr(sb, show_date(date, tz, DATE_RFC2822));
	347	return;
	348	case 'r': /* date, relative */
	349	strbuf_addstr(sb, show_date(date, tz, DATE_RELATIVE));
	350	return;
	351	case 'i': /* date, ISO 8601 */
	352	strbuf_addstr(sb, show_date(date, tz, DATE_ISO8601));
	353	return;
	354	}
93fc05eb JS	355	}
93fc05eb JS	356
cde75e59 RS	357	static void format_commit_item(struct strbuf sb, const char placeholder,
cde75e59 RS	358	void *context)
93fc05eb	359	{
cde75e59	360	const struct commit *commit = context;
93fc05eb	361	struct commit_list *p;
93fc05eb JS	362	int i;
	363	enum { HEADER, SUBJECT, BODY } state;
	364	const char *msg = commit->buffer;
	365
93fc05eb	366	/* these are independent of the commit */
cde75e59 RS	367	switch (placeholder[0]) {
	368	case 'C':
	369	switch (placeholder[3]) {
	370	case 'd': /* red */
	371	strbuf_addstr(sb, "\033[31m");
	372	return;
	373	case 'e': /* green */
	374	strbuf_addstr(sb, "\033[32m");
	375	return;
	376	case 'u': /* blue */
	377	strbuf_addstr(sb, "\033[34m");
	378	return;
	379	case 's': /* reset color */
	380	strbuf_addstr(sb, "\033[m");
	381	return;
	382	}
	383	case 'n': /* newline */
	384	strbuf_addch(sb, '\n');
	385	return;
	386	}
93fc05eb JS	387
	388	/* these depend on the commit */
	389	if (!commit->object.parsed)
	390	parse_object(commit->object.sha1);
93fc05eb	391
cde75e59 RS	392	switch (placeholder[0]) {
	393	case 'H': /* commit hash */
	394	strbuf_addstr(sb, sha1_to_hex(commit->object.sha1));
	395	return;
	396	case 'h': /* abbreviated commit hash */
	397	strbuf_addstr(sb, find_unique_abbrev(commit->object.sha1,
	398	DEFAULT_ABBREV));
	399	return;
	400	case 'T': /* tree hash */
	401	strbuf_addstr(sb, sha1_to_hex(commit->tree->object.sha1));
	402	return;
	403	case 't': /* abbreviated tree hash */
	404	strbuf_addstr(sb, find_unique_abbrev(commit->tree->object.sha1,
	405	DEFAULT_ABBREV));
	406	return;
	407	case 'P': /* parent hashes */
	408	for (p = commit->parents; p; p = p->next) {
	409	if (p != commit->parents)
	410	strbuf_addch(sb, ' ');
	411	strbuf_addstr(sb, sha1_to_hex(p->item->object.sha1));
	412	}
	413	return;
	414	case 'p': /* abbreviated parent hashes */
	415	for (p = commit->parents; p; p = p->next) {
	416	if (p != commit->parents)
	417	strbuf_addch(sb, ' ');
	418	strbuf_addstr(sb, find_unique_abbrev(
	419	p->item->object.sha1, DEFAULT_ABBREV));
	420	}
	421	return;
	422	case 'm': /* left/right/bottom */
	423	strbuf_addch(sb, (commit->object.flags & BOUNDARY)
	424	? '-'
	425	: (commit->object.flags & SYMMETRIC_LEFT)
	426	? '<'
	427	: '>');
	428	return;
	429	}
	430
	431	/* For the rest we have to parse the commit header. */
93fc05eb JS	432	for (i = 0, state = HEADER; msg[i] && state < BODY; i++) {
	433	int eol;
	434	for (eol = i; msg[eol] && msg[eol] != '\n'; eol++)
	435	; /* do nothing */
	436
	437	if (state == SUBJECT) {
cde75e59 RS	438	if (placeholder[0] == 's') {
	439	strbuf_add(sb, msg + i, eol - i);
	440	return;
	441	}
93fc05eb JS	442	i = eol;
	443	}
	444	if (i == eol) {
	445	state++;
	446	/* strip empty lines */
	447	while (msg[eol + 1] == '\n')
	448	eol++;
cde75e59 RS	449	} else if (!prefixcmp(msg + i, "author ")) {
	450	if (placeholder[0] == 'a') {
	451	format_person_part(sb, placeholder[1],
	452	msg + i + 7, eol - i - 7);
	453	return;
	454	}
	455	} else if (!prefixcmp(msg + i, "committer ")) {
	456	if (placeholder[0] == 'c') {
	457	format_person_part(sb, placeholder[1],
	458	msg + i + 10, eol - i - 10);
	459	return;
	460	}
	461	} else if (!prefixcmp(msg + i, "encoding ")) {
	462	if (placeholder[0] == 'e') {
	463	strbuf_add(sb, msg + i + 9, eol - i - 9);
	464	return;
	465	}
	466	}
93fc05eb JS	467	i = eol;
93fc05eb JS	468	}
cde75e59 RS	469	if (msg[i] && placeholder[0] == 'b') /* body */
	470	strbuf_addstr(sb, msg + i);
	471	}
	472
	473	void format_commit_message(const struct commit *commit,
	474	const void format, struct strbuf sb)
	475	{
	476	const char *placeholders[] = {
	477	"H", /* commit hash */
	478	"h", /* abbreviated commit hash */
	479	"T", /* tree hash */
	480	"t", /* abbreviated tree hash */
	481	"P", /* parent hashes */
	482	"p", /* abbreviated parent hashes */
	483	"an", /* author name */
	484	"ae", /* author email */
	485	"ad", /* author date */
	486	"aD", /* author date, RFC2822 style */
	487	"ar", /* author date, relative */
	488	"at", /* author date, UNIX timestamp */
	489	"ai", /* author date, ISO 8601 */
	490	"cn", /* committer name */
	491	"ce", /* committer email */
	492	"cd", /* committer date */
	493	"cD", /* committer date, RFC2822 style */
	494	"cr", /* committer date, relative */
	495	"ct", /* committer date, UNIX timestamp */
	496	"ci", /* committer date, ISO 8601 */
	497	"e", /* encoding */
	498	"s", /* subject */
	499	"b", /* body */
	500	"Cred", /* red */
	501	"Cgreen", /* green */
	502	"Cblue", /* blue */
	503	"Creset", /* reset color */
	504	"n", /* newline */
	505	"m", /* left/right/bottom */
	506	NULL
	507	};
	508	strbuf_expand(sb, format, placeholders, format_commit_item, (void *)commit);
93fc05eb JS	509	}
	510
	511	static void pp_header(enum cmit_fmt fmt,
	512	int abbrev,
	513	enum date_mode dmode,
	514	const char *encoding,
	515	const struct commit *commit,
	516	const char **msg_p,
	517	struct strbuf *sb)
	518	{
	519	int parents_shown = 0;
	520
	521	for (;;) {
	522	const char line = msg_p;
	523	int linelen = get_one_line(*msg_p);
	524
	525	if (!linelen)
	526	return;
	527	*msg_p += linelen;
	528
	529	if (linelen == 1)
	530	/* End of header */
	531	return;
	532
	533	if (fmt == CMIT_FMT_RAW) {
	534	strbuf_add(sb, line, linelen);
	535	continue;
	536	}
	537
	538	if (!memcmp(line, "parent ", 7)) {
	539	if (linelen != 48)
	540	die("bad parent line in commit");
	541	continue;
	542	}
	543
	544	if (!parents_shown) {
	545	struct commit_list *parent;
	546	int num;
	547	for (parent = commit->parents, num = 0;
	548	parent;
	549	parent = parent->next, num++)
	550	;
	551	/* with enough slop */
	552	strbuf_grow(sb, num * 50 + 20);
	553	add_merge_info(fmt, sb, commit, abbrev);
	554	parents_shown = 1;
	555	}
	556
	557	/*
	558	* MEDIUM == DEFAULT shows only author with dates.
	559	* FULL shows both authors but not dates.
	560	* FULLER shows both authors and dates.
	561	*/
	562	if (!memcmp(line, "author ", 7)) {
	563	strbuf_grow(sb, linelen + 80);
	564	add_user_info("Author", fmt, sb, line + 7, dmode, encoding);
	565	}
	566	if (!memcmp(line, "committer ", 10) &&
	567	(fmt == CMIT_FMT_FULL \|\| fmt == CMIT_FMT_FULLER)) {
	568	strbuf_grow(sb, linelen + 80);
	569	add_user_info("Commit", fmt, sb, line + 10, dmode, encoding);
	570	}
	571	}
	572	}
573
574	static void pp_title_line(enum cmit_fmt fmt,
575	const char **msg_p,
576	struct strbuf *sb,
577	const char *subject,
578	const char *after_subject,
579	const char *encoding,
580	int plain_non_ascii)
581	{
582	struct strbuf title;
583
584	strbuf_init(&title, 80);
585
586	for (;;) {
587	const char line = msg_p;
588	int linelen = get_one_line(line);
589
590	*msg_p += linelen;
591	if (!linelen \|\| is_empty_line(line, &linelen))
592	break;
593
594	strbuf_grow(&title, linelen + 2);
595	if (title.len) {
596	if (fmt == CMIT_FMT_EMAIL) {
597	strbuf_addch(&title, '\n');
598	}
599	strbuf_addch(&title, ' ');
600	}
601	strbuf_add(&title, line, linelen);
602	}
603
604	strbuf_grow(sb, title.len + 1024);
605	if (subject) {
606	strbuf_addstr(sb, subject);
607	add_rfc2047(sb, title.buf, title.len, encoding);
608	} else {
609	strbuf_addbuf(sb, &title);
610	}
611	strbuf_addch(sb, '\n');
612
613	if (plain_non_ascii) {
614	const char *header_fmt =
615	"MIME-Version: 1.0\n"
616	"Content-Type: text/plain; charset=%s\n"
617	"Content-Transfer-Encoding: 8bit\n";
618	strbuf_addf(sb, header_fmt, encoding);
619	}
620	if (after_subject) {
621	strbuf_addstr(sb, after_subject);
622	}
623	if (fmt == CMIT_FMT_EMAIL) {
624	strbuf_addch(sb, '\n');
625	}
626	strbuf_release(&title);
627	}
628
629	static void pp_remainder(enum cmit_fmt fmt,
630	const char **msg_p,
631	struct strbuf *sb,
632	int indent)
633	{
634	int first = 1;
635	for (;;) {
636	const char line = msg_p;
637	int linelen = get_one_line(line);
638	*msg_p += linelen;
639
640	if (!linelen)
641	break;
642
643	if (is_empty_line(line, &linelen)) {
644	if (first)
645	continue;
646	if (fmt == CMIT_FMT_SHORT)
647	break;
648	}
649	first = 0;
650
651	strbuf_grow(sb, linelen + indent + 20);
652	if (indent) {
653	memset(sb->buf + sb->len, ' ', indent);
654	strbuf_setlen(sb, sb->len + indent);
655	}
656	strbuf_add(sb, line, linelen);
657	strbuf_addch(sb, '\n');
658	}
659	}
660
661	void pretty_print_commit(enum cmit_fmt fmt, const struct commit *commit,
662	struct strbuf *sb, int abbrev,
663	const char subject, const char after_subject,
664	enum date_mode dmode, int plain_non_ascii)
665	{
666	unsigned long beginning_of_body;
667	int indent = 4;
668	const char *msg = commit->buffer;
669	char *reencoded;
670	const char *encoding;
671
672	if (fmt == CMIT_FMT_USERFORMAT) {
673	format_commit_message(commit, user_format, sb);
674	return;
675	}
676
677	encoding = (git_log_output_encoding
678	? git_log_output_encoding
679	: git_commit_encoding);
680	if (!encoding)
681	encoding = "utf-8";
682	reencoded = logmsg_reencode(commit, encoding);
683	if (reencoded) {
684	msg = reencoded;
685	}
686
687	if (fmt == CMIT_FMT_ONELINE \|\| fmt == CMIT_FMT_EMAIL)
688	indent = 0;
689
690	/* After-subject is used to pass in Content-Type: multipart
691	* MIME header; in that case we do not have to do the
692	* plaintext content type even if the commit message has
693	* non 7-bit ASCII character. Otherwise, check if we need
694	* to say this is not a 7-bit ASCII.
695	*/
696	if (fmt == CMIT_FMT_EMAIL && !after_subject) {
697	int i, ch, in_body;
698
699	for (in_body = i = 0; (ch = msg[i]); i++) {
700	if (!in_body) {
701	/* author could be non 7-bit ASCII but
702	* the log may be so; skip over the
703	* header part first.
704	*/
705	if (ch == '\n' && msg[i+1] == '\n')
706	in_body = 1;
707	}
708	else if (non_ascii(ch)) {
709	plain_non_ascii = 1;
710	break;
711	}
712	}
713	}
714
715	pp_header(fmt, abbrev, dmode, encoding, commit, &msg, sb);
716	if (fmt != CMIT_FMT_ONELINE && !subject) {
717	strbuf_addch(sb, '\n');
718	}
719
720	/* Skip excess blank lines at the beginning of body, if any... */
721	for (;;) {
722	int linelen = get_one_line(msg);
723	int ll = linelen;
724	if (!linelen)
725	break;
726	if (!is_empty_line(msg, &ll))
727	break;
728	msg += linelen;
729	}
730
731	/* These formats treat the title line specially. */
732	if (fmt == CMIT_FMT_ONELINE \|\| fmt == CMIT_FMT_EMAIL)
733	pp_title_line(fmt, &msg, sb, subject,
734	after_subject, encoding, plain_non_ascii);
735
736	beginning_of_body = sb->len;
737	if (fmt != CMIT_FMT_ONELINE)
738	pp_remainder(fmt, &msg, sb, indent);
739	strbuf_rtrim(sb);
740
741	/* Make sure there is an EOLN for the non-oneline case */
742	if (fmt != CMIT_FMT_ONELINE)
743	strbuf_addch(sb, '\n');
744
745	/*
746	* The caller may append additional body text in e-mail
747	* format. Make sure we did not strip the blank line
748	* between the header and the body.
749	*/
750	if (fmt == CMIT_FMT_EMAIL && sb->len <= beginning_of_body)
751	strbuf_addch(sb, '\n');
752	free(reencoded);
753	}