git/list[1] front-page[2] threads[3] people[4] search[5] about
 

[PATCH] Teach "git add" and friends to be paranoid

From
Junio C Hamano <gitster@pobox.com>
Date
Feb 18, 2010, 01:16 UTC
Message-ID
<7vljer1gyg.fsf_-_@alter.siamese.dyndns.org>
In-Reply-To
<20100214011812.GA2175@dpotapov.dyndns.org>

When creating a loose object, we normally mmap(2) the entire file, and hash and then compress to write it out in two separate steps for efficiency.

This is perfectly good for the intended use of git---nobody is supposed to be insane enough to expect that it won't break anything to muck with the contents of a file after telling git to index it and before getting the control back from git.

But the nature of breakage caused by such an abuse is rather bad. We will end up with loose object files, whose names do not match what are stored and recovered when uncompressed.

This teaches the index_mem() codepath to be paranoid and hash and compress the data after reading it in core. The contents hashed may not match the contents of the file in an insane use case, but at least this way the result will be internally consistent.

Signed-off-by: Junio C Hamano <gitster@pobox.com>
---
 sha1_file.c |   81 ++++++++++++++++++++++++++++++++++++++++++++++++-----------
 1 files changed, 66 insertions(+), 15 deletions(-)
diff --git a/sha1_file.c b/sha1_file.c
index 657825e..d8a7722 100644
--- a/sha1_file.c
+++ b/sha1_file.c
@@ -2278,7 +2278,8 @@ static int create_tmpfile(char *buffer, size_t bufsiz, const char *filename)
 }
 
 static int write_loose_object(const unsigned char *sha1, char *hdr, int hdrlen,
-			      void *buf, unsigned long len, time_t mtime)
+			      void *buf, unsigned long len, time_t mtime,
+			      int paranoid)
 {
 	int fd, ret;
 	size_t size;
@@ -2286,6 +2287,7 @@ static int write_loose_object(const unsigned char *sha1, char *hdr, int hdrlen,
 	z_stream stream;
 	char *filename;
 	static char tmpfile[PATH_MAX];
+	git_SHA_CTX ctx;
 
 	filename = sha1_file_name(sha1);
 	fd = create_tmpfile(tmpfile, sizeof(tmpfile), filename);
@@ -2312,12 +2314,41 @@ static int write_loose_object(const unsigned char *sha1, char *hdr, int hdrlen,
 	stream.next_in = (unsigned char *)hdr;
 	stream.avail_in = hdrlen;
 	while (deflate(&stream, 0) == Z_OK)
-		/* nothing */;
+		; /* nothing */
 
 	/* Then the data itself.. */
-	stream.next_in = buf;
-	stream.avail_in = len;
-	ret = deflate(&stream, Z_FINISH);
+	if (paranoid) {
+		unsigned char stablebuf[262144];
+		char *bufptr = buf;
+		unsigned long remainder = len;
+
+		git_SHA1_Init(&ctx);
+		git_SHA1_Update(&ctx, hdr, hdrlen);
+
+		ret = Z_OK;
+		while (remainder) {
+			unsigned long chunklen = remainder;
+
+			if (sizeof(stablebuf) <= chunklen)
+				chunklen = sizeof(stablebuf);
+			memcpy(stablebuf, bufptr, chunklen);
+			git_SHA1_Update(&ctx, stablebuf, chunklen);
+			stream.next_in = stablebuf;
+			stream.avail_in = chunklen;
+			do {
+				ret = deflate(&stream, Z_NO_FLUSH);
+			} while (ret == Z_OK);
+			bufptr += chunklen;
+			remainder -= chunklen;
+		}
+		if (ret != Z_STREAM_END)
+			ret = deflate(&stream, Z_FINISH);
+	} else {
+		stream.next_in = buf;
+		stream.avail_in = len;
+		ret = deflate(&stream, Z_FINISH);
+	}
+
 	if (ret != Z_STREAM_END)
 		die("unable to deflate new object %s (%d)", sha1_to_hex(sha1), ret);
 
@@ -2327,6 +2358,12 @@ static int write_loose_object(const unsigned char *sha1, char *hdr, int hdrlen,
 
 	size = stream.total_out;
 
+	if (paranoid) {
+		unsigned char paranoid_sha1[20];
+		git_SHA1_Final(paranoid_sha1, &ctx);
+		if (hashcmp(paranoid_sha1, sha1))
+			die("hashed file is volatile");
+	}
 	if (write_buffer(fd, compressed, size) < 0)
 		die("unable to write sha1 file");
 	close_sha1_file(fd);
@@ -2344,7 +2381,7 @@ static int write_loose_object(const unsigned char *sha1, char *hdr, int hdrlen,
 	return move_temp_to_file(tmpfile, filename);
 }
 
-int write_sha1_file(void *buf, unsigned long len, const char *type, unsigned char *returnsha1)
+static int write_sha1_file_paranoid(void *buf, unsigned long len, const char *type, unsigned char *returnsha1, int paranoid)
 {
 	unsigned char sha1[20];
 	char hdr[32];
@@ -2358,7 +2395,12 @@ int write_sha1_file(void *buf, unsigned long len, const char *type, unsigned cha
 		hashcpy(returnsha1, sha1);
 	if (has_sha1_file(sha1))
 		return 0;
-	return write_loose_object(sha1, hdr, hdrlen, buf, len, 0);
+	return write_loose_object(sha1, hdr, hdrlen, buf, len, 0, paranoid);
+}
+
+int write_sha1_file(void *buf, unsigned long len, const char *type, unsigned char *returnsha1)
+{
+	return write_sha1_file_paranoid(buf, len, type, returnsha1, 0);
 }
 
 int force_object_loose(const unsigned char *sha1, time_t mtime)
@@ -2376,7 +2418,7 @@ int force_object_loose(const unsigned char *sha1, time_t mtime)
 	if (!buf)
 		return error("cannot read sha1_file for %s", sha1_to_hex(sha1));
 	hdrlen = sprintf(hdr, "%s %lu", typename(type), len) + 1;
-	ret = write_loose_object(sha1, hdr, hdrlen, buf, len, mtime);
+	ret = write_loose_object(sha1, hdr, hdrlen, buf, len, mtime, 0);
 	free(buf);
 
 	return ret;
@@ -2405,10 +2447,15 @@ int has_sha1_file(const unsigned char *sha1)
 	return has_loose_object(sha1);
 }
 
+#define INDEX_MEM_WRITE_OBJECT  01
+#define INDEX_MEM_PARANOID      02
+
 static int index_mem(unsigned char *sha1, void *buf, size_t size,
-		     int write_object, enum object_type type, const char *path)
+		     enum object_type type, const char *path, int flag)
 {
 	int ret, re_allocated = 0;
+	int write_object = flag & INDEX_MEM_WRITE_OBJECT;
+	int paranoid = flag & INDEX_MEM_PARANOID;
 
 	if (!type)
 		type = OBJ_BLOB;
@@ -2426,9 +2473,11 @@ static int index_mem(unsigned char *sha1, void *buf, size_t size,
 	}
 
 	if (write_object)
-		ret = write_sha1_file(buf, size, typename(type), sha1);
+		ret = write_sha1_file_paranoid(buf, size, typename(type),
+					       sha1, paranoid);
 	else
 		ret = hash_sha1_file(buf, size, typename(type), sha1);
+
 	if (re_allocated)
 		free(buf);
 	return ret;
@@ -2437,23 +2486,25 @@ static int index_mem(unsigned char *sha1, void *buf, size_t size,
 int index_fd(unsigned char *sha1, int fd, struct stat *st, int write_object,
 	     enum object_type type, const char *path)
 {
-	int ret;
+	int ret, flag;
 	size_t size = xsize_t(st->st_size);
 
+	flag = write_object ? INDEX_MEM_WRITE_OBJECT : 0;
 	if (!S_ISREG(st->st_mode)) {
 		struct strbuf sbuf = STRBUF_INIT;
 		if (strbuf_read(&sbuf, fd, 4096) >= 0)
-			ret = index_mem(sha1, sbuf.buf, sbuf.len, write_object,
-					type, path);
+			ret = index_mem(sha1, sbuf.buf, sbuf.len,
+					type, path, flag);
 		else
 			ret = -1;
 		strbuf_release(&sbuf);
 	} else if (size) {
 		void *buf = xmmap(NULL, size, PROT_READ, MAP_PRIVATE, fd, 0);
-		ret = index_mem(sha1, buf, size, write_object, type, path);
+		flag |= INDEX_MEM_PARANOID;
+		ret = index_mem(sha1, buf, size, type, path, flag);
 		munmap(buf, size);
 	} else
-		ret = index_mem(sha1, NULL, size, write_object, type, path);
+		ret = index_mem(sha1, NULL, size, type, path, flag);
 	close(fd);
 	return ret;
 }
-- 
1.7.0.81.g58679
Previous: Dmitry PotapovNext: Junio C Hamano
Message 38 of 84 in “Re: Bug#569505: git-core: 'git add' corrupts repository if the working directory is modified as it runs”
  1. Jonathan NiederFeb 12, 2010
  2. Zygo BlaxellFeb 12, 2010
  3. Jonathan NiederFeb 13, 2010
  4. Ilari LiusvaaraFeb 13, 2010
  5. Thomas RastFeb 13, 2010
  6. Ilari LiusvaaraFeb 13, 2010
  7. Dmitry PotapovFeb 13, 2010
  8. Zygo BlaxellFeb 13, 2010
  9. don't use mmap() to hash filesDmitry Potapov, Feb 14, 2010
  10. Junio C HamanoFeb 14, 2010
  11. Dmitry PotapovFeb 14, 2010
  12. Junio C HamanoFeb 14, 2010
  13. Thomas RastFeb 14, 2010
  14. Junio C HamanoFeb 14, 2010
  15. Johannes SchindelinFeb 14, 2010
  16. Junio C HamanoFeb 14, 2010
  17. Dmitry PotapovFeb 14, 2010
  18. Jakub NarebskiFeb 14, 2010
  19. Paolo BonziniFeb 14, 2010
  20. Johannes SchindelinFeb 14, 2010
  21. Dmitry PotapovFeb 14, 2010
  22. Johannes SchindelinFeb 14, 2010
  23. Johannes SchindelinFeb 14, 2010
  24. Dmitry PotapovFeb 14, 2010
  25. Zygo BlaxellFeb 14, 2010
  26. Nicolas PitreFeb 15, 2010
  27. Dmitry PotapovFeb 15, 2010
  28. Paolo BonziniFeb 15, 2010
  29. Dmitry PotapovFeb 15, 2010
  30. Dmitry PotapovFeb 14, 2010
  31. Avery PennarunFeb 14, 2010
  32. Nicolas PitreFeb 15, 2010
  33. Avery PennarunFeb 15, 2010
  34. Nicolas PitreFeb 15, 2010
  35. Avery PennarunFeb 15, 2010
  36. Nicolas PitreFeb 15, 2010
  37. don't use mmap() to hash filesDmitry Potapov, Feb 14, 2010
  38. Teach "git add" and friends to be paranoidJunio C Hamano, Feb 18, 2010
  39. Junio C HamanoFeb 18, 2010
  40. Zygo BlaxellFeb 18, 2010
  41. Junio C HamanoFeb 19, 2010
  42. Jeff KingFeb 18, 2010
  43. Nicolas PitreFeb 18, 2010
  44. Junio C HamanoFeb 18, 2010
  45. Wincent ColaiutaFeb 18, 2010
  46. Zygo BlaxellFeb 18, 2010
  47. Jonathan NiederFeb 18, 2010
  48. Junio C HamanoFeb 18, 2010
  49. Paolo BonziniFeb 22, 2010
  50. Dmitry PotapovFeb 22, 2010
  51. Thomas RastFeb 18, 2010
  52. Junio C HamanoFeb 18, 2010
  53. Nicolas PitreFeb 18, 2010
  54. 16 gig, 350,000 file repositoryBill Lear, Feb 18, 2010
  55. Nicolas PitreFeb 18, 2010
  56. Erik Faye-LundFeb 19, 2010
  57. Bill LearFeb 22, 2010
  58. Nicolas PitreFeb 22, 2010
  59. Peter HarrisFeb 18, 2010
  60. Junio C HamanoFeb 18, 2010
  61. Nicolas PitreFeb 18, 2010
  62. Jonathan NiederFeb 19, 2010
  63. Zygo BlaxellFeb 19, 2010
  64. Junio C HamanoFeb 19, 2010
  65. Zygo BlaxellFeb 19, 2010
  66. Dmitry PotapovFeb 19, 2010
  67. Junio C HamanoFeb 19, 2010
  68. Junio C HamanoFeb 20, 2010
  69. Dmitry PotapovFeb 21, 2010
  70. Junio C HamanoFeb 21, 2010
  71. Dmitry PotapovFeb 22, 2010
  72. Junio C HamanoFeb 22, 2010
  73. Dmitry PotapovFeb 22, 2010
  74. Nicolas PitreFeb 22, 2010
  75. Dmitry PotapovFeb 22, 2010
  76. Zygo BlaxellFeb 22, 2010
  77. Nicolas PitreFeb 22, 2010
  78. Junio C HamanoFeb 22, 2010
  79. Nicolas PitreFeb 22, 2010
  80. Dmitry PotapovFeb 22, 2010
  81. Nicolas PitreFeb 22, 2010
  82. mmap with MAP_PRIVATE is useless (was Re: Bug#569505: git-core: 'git add' corrupts repository if the working directory is modified as it runs)Paolo Bonzini, Feb 14, 2010
  83. Junio C HamanoFeb 14, 2010
  84. Paolo BonziniFeb 14, 2010

Read the whole thread, see it on lore, or plain text.

$ cat FOOTERMessages come from the public archive at lore.kernel.org/git, fetched every hour. The front page is chosen and written each morning by an AI editor and can be wrong; the threads themselves are the record. About and API. For agents: an MCP server at https://gitlist.dev/mcp, and any thread, story or person page as Markdown by adding .md to its URL (or sending Accept: text/markdown). Details in /llms.txt.