git/list[1] front-page[2] threads[3] people[4] search[5] about
 

[PATCH v2 07/10] index-pack: reduce memory usage when the pack has large blobs

From
Nguyễn Thái Ngọc Duy <pclouds@gmail.com>
Date
Mar 4, 2012, 12:59 UTC
Message-ID
<1330865996-2069-8-git-send-email-pclouds@gmail.com>
In-Reply-To
<1330865996-2069-1-git-send-email-pclouds@gmail.com>
This command unpacks every non-delta objects in order to:
1. calculate sha-1
2. do byte-to-byte sha-1 collision test if we happen to have objects
   with the same sha-1
3. validate object content in strict mode

All this requires the entire object to stay in memory, a bad news for giant blobs. This patch lowers memory consumption by not saving the object in memory whenever possible, calculating SHA-1 while unpacking the object.

This patch assumes that the collision test is rarely needed. The collision test will be done later in second pass if necessary, which puts the entire object back to memory again (We could even do the collision test without putting the entire object back in memory, by comparing as we unpack it).

In strict mode, it always keeps non-blob objects in memory for validation (blobs do not need data validation).

Signed-off-by: Nguyễn Thái Ngọc Duy <pclouds@gmail.com>
---
 builtin/index-pack.c |   64 +++++++++++++++++++++++++++++++++++++++----------
 t/t1050-large.sh     |    4 +-
 2 files changed, 53 insertions(+), 15 deletions(-)
diff --git a/builtin/index-pack.c b/builtin/index-pack.c
index 918684f..db27133 100644
--- a/builtin/index-pack.c
+++ b/builtin/index-pack.c
@@ -276,30 +276,60 @@ static void unlink_base_data(struct base_data *c)
 	free_base_data(c);
 }
 
-static void *unpack_entry_data(unsigned long offset, unsigned long size)
+static void *unpack_entry_data(unsigned long offset, unsigned long size,
+			       enum object_type type, unsigned char *sha1)
 {
+	static char fixed_buf[8192];
 	int status;
 	git_zstream stream;
-	void *buf = xmalloc(size);
+	void *buf;
+	git_SHA_CTX c;
+
+	if (sha1) {		/* do hash_sha1_file internally */
+		char hdr[32];
+		int hdrlen = sprintf(hdr, "%s %lu", typename(type), size)+1;
+		git_SHA1_Init(&c);
+		git_SHA1_Update(&c, hdr, hdrlen);
+
+		buf = fixed_buf;
+	} else {
+		buf = xmalloc(size);
+	}
 
 	memset(&stream, 0, sizeof(stream));
 	git_inflate_init(&stream);
 	stream.next_out = buf;
-	stream.avail_out = size;
+	stream.avail_out = buf == fixed_buf ? sizeof(fixed_buf) : size;
 
 	do {
 		stream.next_in = fill(1);
 		stream.avail_in = input_len;
 		status = git_inflate(&stream, 0);
 		use(input_len - stream.avail_in);
+		if (sha1) {
+			git_SHA1_Update(&c, buf, stream.next_out - (unsigned char *)buf);
+			stream.next_out = buf;
+			stream.avail_out = sizeof(fixed_buf);
+		}
 	} while (status == Z_OK);
 	if (stream.total_out != size || status != Z_STREAM_END)
 		bad_object(offset, "inflate returned %d", status);
 	git_inflate_end(&stream);
+	if (sha1) {
+		git_SHA1_Final(sha1, &c);
+		buf = NULL;
+	}
 	return buf;
 }
 
-static void *unpack_raw_entry(struct object_entry *obj, union delta_base *delta_base)
+static int is_delta_type(enum object_type type)
+{
+	return (type == OBJ_REF_DELTA || type == OBJ_OFS_DELTA);
+}
+
+static void *unpack_raw_entry(struct object_entry *obj,
+			      union delta_base *delta_base,
+			      unsigned char *sha1)
 {
 	unsigned char *p;
 	unsigned long size, c;
@@ -359,7 +389,9 @@ static void *unpack_raw_entry(struct object_entry *obj, union delta_base *delta_
 	}
 	obj->hdr_size = consumed_bytes - obj->idx.offset;
 
-	data = unpack_entry_data(obj->idx.offset, obj->size);
+	if (is_delta_type(obj->type) || strict)
+		sha1 = NULL;	/* save unpacked object */
+	data = unpack_entry_data(obj->idx.offset, obj->size, obj->type, sha1);
 	obj->idx.crc32 = input_crc32;
 	return data;
 }
@@ -460,8 +492,9 @@ static void find_delta_children(const union delta_base *base,
 static void sha1_object(const void *data, unsigned long size,
 			enum object_type type, unsigned char *sha1)
 {
-	hash_sha1_file(data, size, typename(type), sha1);
-	if (has_sha1_file(sha1)) {
+	if (data)
+		hash_sha1_file(data, size, typename(type), sha1);
+	if (data && has_sha1_file(sha1)) {
 		void *has_data;
 		enum object_type has_type;
 		unsigned long has_size;
@@ -510,11 +543,6 @@ static void sha1_object(const void *data, unsigned long size,
 	}
 }
 
-static int is_delta_type(enum object_type type)
-{
-	return (type == OBJ_REF_DELTA || type == OBJ_OFS_DELTA);
-}
-
 /*
  * This function is part of find_unresolved_deltas(). There are two
  * walkers going in the opposite ways.
@@ -689,10 +717,20 @@ static int compare_delta_entry(const void *a, const void *b)
  * - if used as a base, uncompress the object and apply all deltas,
  *   recursively checking if the resulting object is used as a base
  *   for some more deltas.
+ * - if the same object exists in repository and we're not in strict
+ *   mode, we skipped the sha-1 collision test in the first pass.
+ *   Do it now.
  */
 static void second_pass(struct object_entry *obj)
 {
 	struct base_data *base_obj = alloc_base_data();
+
+	if (!strict && has_sha1_file(obj->idx.sha1)) {
+		void *data = get_data_from_pack(obj);
+		sha1_object(data, obj->size, obj->type, obj->idx.sha1);
+		free(data);
+	}
+
 	base_obj->obj = obj;
 	base_obj->data = NULL;
 	find_unresolved_deltas(base_obj);
@@ -718,7 +756,7 @@ static void parse_pack_objects(unsigned char *sha1)
 				nr_objects);
 	for (i = 0; i < nr_objects; i++) {
 		struct object_entry *obj = &objects[i];
-		void *data = unpack_raw_entry(obj, &delta->base);
+		void *data = unpack_raw_entry(obj, &delta->base, obj->idx.sha1);
 		obj->real_type = obj->type;
 		if (is_delta_type(obj->type)) {
 			nr_deltas++;
diff --git a/t/t1050-large.sh b/t/t1050-large.sh
index 66acb3b..7e78c72 100755
--- a/t/t1050-large.sh
+++ b/t/t1050-large.sh
@@ -123,7 +123,7 @@ test_expect_success 'git-show a large file' '
 
 '
 
-test_expect_failure 'clone' '
+test_expect_success 'clone' '
 	git clone -n file://"$PWD"/.git new &&
 	(
 	cd new &&
@@ -132,7 +132,7 @@ test_expect_failure 'clone' '
 	)
 '
 
-test_expect_failure 'fetch updates' '
+test_expect_success 'fetch updates' '
 	echo modified >> large1 &&
 	git commit -q -a -m updated &&
 	(
-- 
1.7.8.36.g69ee2
Previous: Nguyễn Thái Ngọc DuyNext: Nguyễn Thái Ngọc Duy
Message 31 of 48 in “Large blob fixes”
  1. 00/11 Large blob fixesNguyễn Thái Ngọc Duy, Feb 27, 2012
  2. 01/11 Add more large blob test casesNguyễn Thái Ngọc Duy, Feb 27, 2012
  3. Peter BaumannFeb 27, 2012
  4. 02/11 Factor out and export large blob writing code to arbitrary file handleNguyễn Thái Ngọc Duy, Feb 27, 2012
  5. Junio C HamanoFeb 27, 2012
  6. Junio C HamanoFeb 27, 2012
  7. 03/11 cat-file: use streaming interface to print blobsNguyễn Thái Ngọc Duy, Feb 27, 2012
  8. Junio C HamanoFeb 27, 2012
  9. Nguyen Thai Ngoc DuyFeb 28, 2012
  10. 04/11 parse_object: special code path for blobs to avoid putting whole object in memoryNguyễn Thái Ngọc Duy, Feb 27, 2012
  11. 05/11 show: use streaming interface for showing blobsNguyễn Thái Ngọc Duy, Feb 27, 2012
  12. Junio C HamanoFeb 27, 2012
  13. 06/11 index-pack --verify: skip sha-1 collision testNguyễn Thái Ngọc Duy, Feb 27, 2012
  14. 07/11 index-pack: split second pass obj handling into own functionNguyễn Thái Ngọc Duy, Feb 27, 2012
  15. 08/11 index-pack: reduce memory usage when the pack has large blobsNguyễn Thái Ngọc Duy, Feb 27, 2012
  16. 09/11 pack-check: do not unpack blobsNguyễn Thái Ngọc Duy, Feb 27, 2012
  17. 10/11 archive: support streaming large files to a tar archiveNguyễn Thái Ngọc Duy, Feb 27, 2012
  18. 11/11 fsck: use streaming interface for writing lost-found blobsNguyễn Thái Ngọc Duy, Feb 27, 2012
  19. Junio C HamanoFeb 27, 2012
  20. Nguyen Thai Ngoc DuyFeb 28, 2012
  21. 00/10 Large blob fixesNguyễn Thái Ngọc Duy, Mar 4, 2012
  22. 01/10 Add more large blob test casesNguyễn Thái Ngọc Duy, Mar 4, 2012
  23. Junio C HamanoMar 6, 2012
  24. 02/10 streaming: make streaming-write-entry to be more reusableNguyễn Thái Ngọc Duy, Mar 4, 2012
  25. 03/10 cat-file: use streaming interface to print blobsNguyễn Thái Ngọc Duy, Mar 4, 2012
  26. Junio C HamanoMar 4, 2012
  27. Nguyen Thai Ngoc DuyMar 5, 2012
  28. 04/10 parse_object: special code path for blobs to avoid putting whole object in memoryNguyễn Thái Ngọc Duy, Mar 4, 2012
  29. 05/10 show: use streaming interface for showing blobsNguyễn Thái Ngọc Duy, Mar 4, 2012
  30. 06/10 index-pack: split second pass obj handling into own functionNguyễn Thái Ngọc Duy, Mar 4, 2012
  31. 07/10 index-pack: reduce memory usage when the pack has large blobsNguyễn Thái Ngọc Duy, Mar 4, 2012
  32. 08/10 pack-check: do not unpack blobsNguyễn Thái Ngọc Duy, Mar 4, 2012
  33. 09/10 archive: support streaming large files to a tar archiveNguyễn Thái Ngọc Duy, Mar 4, 2012
  34. 10/10 fsck: use streaming interface for writing lost-found blobsNguyễn Thái Ngọc Duy, Mar 4, 2012
  35. 00/11 Large blob fixesNguyễn Thái Ngọc Duy, Mar 5, 2012
  36. 01/11 Add more large blob test casesNguyễn Thái Ngọc Duy, Mar 5, 2012
  37. 02/11 streaming: make streaming-write-entry to be more reusableNguyễn Thái Ngọc Duy, Mar 5, 2012
  38. 03/11 cat-file: use streaming interface to print blobsNguyễn Thái Ngọc Duy, Mar 5, 2012
  39. 04/11 parse_object: special code path for blobs to avoid putting whole object in memoryNguyễn Thái Ngọc Duy, Mar 5, 2012
  40. Junio C HamanoMar 6, 2012
  41. 05/11 show: use streaming interface for showing blobsNguyễn Thái Ngọc Duy, Mar 5, 2012
  42. 06/11 index-pack: split second pass obj handling into own functionNguyễn Thái Ngọc Duy, Mar 5, 2012
  43. 07/11 index-pack: reduce memory usage when the pack has large blobsNguyễn Thái Ngọc Duy, Mar 5, 2012
  44. 08/11 pack-check: do not unpack blobsNguyễn Thái Ngọc Duy, Mar 5, 2012
  45. 09/11 archive: support streaming large files to a tar archiveNguyễn Thái Ngọc Duy, Mar 5, 2012
  46. Junio C HamanoMar 6, 2012
  47. 10/11 fsck: use streaming interface for writing lost-found blobsNguyễn Thái Ngọc Duy, Mar 5, 2012
  48. 11/11 update-server-info: respect core.bigfilethresholdNguyễn Thái Ngọc Duy, Mar 5, 2012

Read the whole thread, see it on lore, or plain text.

$ cat FOOTERMessages come from the public archive at lore.kernel.org/git, fetched every hour. The front page is chosen and written each morning by an AI editor and can be wrong; the threads themselves are the record. About and API. For agents: an MCP server at https://gitlist.dev/mcp, and any thread, story or person page as Markdown by adding .md to its URL (or sending Accept: text/markdown). Details in /llms.txt.