git/list[1] front-page[2] threads[3] people[4] search[5] about
 

[PATCH v2 07/30] object-file: update the loose object map when writing loose objects

From
EBEric W. Biederman <ebiederm@gmail.com>
Date
Oct 2, 2023, 02:40 UTC
Message-ID
<20231002024034.2611-7-ebiederm@gmail.com>
In-Reply-To
<878r8l929e.fsf@gmail.froward.int.ebiederm.org>
From: "Eric W. Biederman" <ebiederm@xmission.com>

To implement SHA1 compatibility on SHA256 repositories the loose object map needs to be updated whenver a loose object is written. Updating the loose object map this way allows git to support the old hash algorithm in constant time.

The functions write_loose_object, and stream_loose_object are the only two functions that write to the loose object store.

Update stream_loose_object to compute the compatibiilty hash, update the loose object, and then call repo_add_loose_object_map to update the loose object map.

Update write_object_file_flags to convert the object into it's compatibility encoding, hash the compatibility encoding, write the object, and then update the loose object map.

Update force_object_loose to lookup the hash of the compatibility encoding, write the loose object, and then update the loose object map.

Update write_object_file_literally to convert the object into it's compatibility hash encoding, hash the compatibility enconding, write the object, and then update the loose object map, when the type string is a known type. For objects with an unknown type this results in a partially broken repository, as the objects are not mapped.

The point of write_object_file_literally is to generate a partially broken repository for testing. For testing skipping writing the loose object map is much more useful than refusing to write the broken object at all.

Except that the loose objects are updated before the loose object map I have not done any analysis to see how robust this scheme is in the event of failure.

Signed-off-by: "Eric W. Biederman" <ebiederm@xmission.com>
---
 object-file.c | 113 ++++++++++++++++++++++++++++++++++++++++++--------
 1 file changed, 95 insertions(+), 18 deletions(-)
diff --git a/object-file.c b/object-file.c
index 7dc0c4bfbba8..4e55f475b3b4 100644
--- a/object-file.c
+++ b/object-file.c
@@ -43,6 +43,8 @@
 #include "setup.h"
 #include "submodule.h"
 #include "fsck.h"
+#include "loose.h"
+#include "object-file-convert.h"
 
 /* The maximum size for an object header. */
 #define MAX_HEADER_LEN 32
@@ -1952,9 +1954,12 @@ static int start_loose_object_common(struct strbuf *tmp_file,
 				     const char *filename, unsigned flags,
 				     git_zstream *stream,
 				     unsigned char *buf, size_t buflen,
-				     git_hash_ctx *c,
+				     git_hash_ctx *c, git_hash_ctx *compat_c,
 				     char *hdr, int hdrlen)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *algo = repo->hash_algo;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
 	int fd;
 
 	fd = create_tmpfile(tmp_file, filename);
@@ -1974,14 +1979,18 @@ static int start_loose_object_common(struct strbuf *tmp_file,
 	git_deflate_init(stream, zlib_compression_level);
 	stream->next_out = buf;
 	stream->avail_out = buflen;
-	the_hash_algo->init_fn(c);
+	algo->init_fn(c);
+	if (compat && compat_c)
+		compat->init_fn(compat_c);
 
 	/*  Start to feed header to zlib stream */
 	stream->next_in = (unsigned char *)hdr;
 	stream->avail_in = hdrlen;
 	while (git_deflate(stream, 0) == Z_OK)
 		; /* nothing */
-	the_hash_algo->update_fn(c, hdr, hdrlen);
+	algo->update_fn(c, hdr, hdrlen);
+	if (compat && compat_c)
+		compat->update_fn(compat_c, hdr, hdrlen);
 
 	return fd;
 }
@@ -1990,16 +1999,21 @@ static int start_loose_object_common(struct strbuf *tmp_file,
  * Common steps for the inner git_deflate() loop for writing loose
  * objects. Returns what git_deflate() returns.
  */
-static int write_loose_object_common(git_hash_ctx *c,
+static int write_loose_object_common(git_hash_ctx *c, git_hash_ctx *compat_c,
 				     git_zstream *stream, const int flush,
 				     unsigned char *in0, const int fd,
 				     unsigned char *compressed,
 				     const size_t compressed_len)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *algo = repo->hash_algo;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
 	int ret;
 
 	ret = git_deflate(stream, flush ? Z_FINISH : 0);
-	the_hash_algo->update_fn(c, in0, stream->next_in - in0);
+	algo->update_fn(c, in0, stream->next_in - in0);
+	if (compat && compat_c)
+		compat->update_fn(compat_c, in0, stream->next_in - in0);
 	if (write_in_full(fd, compressed, stream->next_out - compressed) < 0)
 		die_errno(_("unable to write loose object file"));
 	stream->next_out = compressed;
@@ -2014,15 +2028,21 @@ static int write_loose_object_common(git_hash_ctx *c,
  * - End the compression of zlib stream.
  * - Get the calculated oid to "oid".
  */
-static int end_loose_object_common(git_hash_ctx *c, git_zstream *stream,
-				   struct object_id *oid)
+static int end_loose_object_common(git_hash_ctx *c, git_hash_ctx *compat_c,
+				   git_zstream *stream, struct object_id *oid,
+				   struct object_id *compat_oid)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *algo = repo->hash_algo;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
 	int ret;
 
 	ret = git_deflate_end_gently(stream);
 	if (ret != Z_OK)
 		return ret;
-	the_hash_algo->final_oid_fn(oid, c);
+	algo->final_oid_fn(oid, c);
+	if (compat && compat_c)
+		compat->final_oid_fn(compat_oid, compat_c);
 
 	return Z_OK;
 }
@@ -2046,7 +2066,7 @@ static int write_loose_object(const struct object_id *oid, char *hdr,
 
 	fd = start_loose_object_common(&tmp_file, filename.buf, flags,
 				       &stream, compressed, sizeof(compressed),
-				       &c, hdr, hdrlen);
+				       &c, NULL, hdr, hdrlen);
 	if (fd < 0)
 		return -1;
 
@@ -2056,14 +2076,14 @@ static int write_loose_object(const struct object_id *oid, char *hdr,
 	do {
 		unsigned char *in0 = stream.next_in;
 
-		ret = write_loose_object_common(&c, &stream, 1, in0, fd,
+		ret = write_loose_object_common(&c, NULL, &stream, 1, in0, fd,
 						compressed, sizeof(compressed));
 	} while (ret == Z_OK);
 
 	if (ret != Z_STREAM_END)
 		die(_("unable to deflate new object %s (%d)"), oid_to_hex(oid),
 		    ret);
-	ret = end_loose_object_common(&c, &stream, &parano_oid);
+	ret = end_loose_object_common(&c, NULL, &stream, &parano_oid, NULL);
 	if (ret != Z_OK)
 		die(_("deflateEnd on object %s failed (%d)"), oid_to_hex(oid),
 		    ret);
@@ -2108,10 +2128,12 @@ static int freshen_packed_object(const struct object_id *oid)
 int stream_loose_object(struct input_stream *in_stream, size_t len,
 			struct object_id *oid)
 {
+	const struct git_hash_algo *compat = the_repository->compat_hash_algo;
+	struct object_id compat_oid;
 	int fd, ret, err = 0, flush = 0;
 	unsigned char compressed[4096];
 	git_zstream stream;
-	git_hash_ctx c;
+	git_hash_ctx c, compat_c;
 	struct strbuf tmp_file = STRBUF_INIT;
 	struct strbuf filename = STRBUF_INIT;
 	int dirlen;
@@ -2135,7 +2157,7 @@ int stream_loose_object(struct input_stream *in_stream, size_t len,
 	 */
 	fd = start_loose_object_common(&tmp_file, filename.buf, 0,
 				       &stream, compressed, sizeof(compressed),
-				       &c, hdr, hdrlen);
+				       &c, &compat_c, hdr, hdrlen);
 	if (fd < 0) {
 		err = -1;
 		goto cleanup;
@@ -2153,7 +2175,7 @@ int stream_loose_object(struct input_stream *in_stream, size_t len,
 			if (in_stream->is_finished)
 				flush = 1;
 		}
-		ret = write_loose_object_common(&c, &stream, flush, in0, fd,
+		ret = write_loose_object_common(&c, &compat_c, &stream, flush, in0, fd,
 						compressed, sizeof(compressed));
 		/*
 		 * Unlike write_loose_object(), we do not have the entire
@@ -2176,7 +2198,7 @@ int stream_loose_object(struct input_stream *in_stream, size_t len,
 	 */
 	if (ret != Z_STREAM_END)
 		die(_("unable to stream deflate new object (%d)"), ret);
-	ret = end_loose_object_common(&c, &stream, oid);
+	ret = end_loose_object_common(&c, &compat_c, &stream, oid, &compat_oid);
 	if (ret != Z_OK)
 		die(_("deflateEnd on stream object failed (%d)"), ret);
 	close_loose_object(fd, tmp_file.buf);
@@ -2203,6 +2225,8 @@ int stream_loose_object(struct input_stream *in_stream, size_t len,
 	}
 
 	err = finalize_object_file(tmp_file.buf, filename.buf);
+	if (!err && compat)
+		err = repo_add_loose_object_map(the_repository, oid, &compat_oid);
 cleanup:
 	strbuf_release(&tmp_file);
 	strbuf_release(&filename);
@@ -2213,17 +2237,38 @@ int write_object_file_flags(const void *buf, unsigned long len,
 			    enum object_type type, struct object_id *oid,
 			    unsigned flags)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *algo = repo->hash_algo;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
+	struct object_id compat_oid;
 	char hdr[MAX_HEADER_LEN];
 	int hdrlen = sizeof(hdr);
 
+	/* Generate compat_oid */
+	if (compat) {
+		if (type == OBJ_BLOB)
+			hash_object_file(compat, buf, len, type, &compat_oid);
+		else {
+			struct strbuf converted = STRBUF_INIT;
+			convert_object_file(&converted, algo, compat,
+					    buf, len, type, 0);
+			hash_object_file(compat, converted.buf, converted.len,
+					 type, &compat_oid);
+			strbuf_release(&converted);
+		}
+	}
+
 	/* Normally if we have it in the pack then we do not bother writing
 	 * it out into .git/objects/??/?{38} file.
 	 */
-	write_object_file_prepare(the_hash_algo, buf, len, type, oid, hdr,
-				  &hdrlen);
+	write_object_file_prepare(algo, buf, len, type, oid, hdr, &hdrlen);
 	if (freshen_packed_object(oid) || freshen_loose_object(oid))
 		return 0;
-	return write_loose_object(oid, hdr, hdrlen, buf, len, 0, flags);
+	if (write_loose_object(oid, hdr, hdrlen, buf, len, 0, flags))
+		return -1;
+	if (compat)
+		return repo_add_loose_object_map(repo, oid, &compat_oid);
+	return 0;
 }
 
 int write_object_file_literally(const void *buf, unsigned long len,
@@ -2231,7 +2276,27 @@ int write_object_file_literally(const void *buf, unsigned long len,
 				unsigned flags)
 {
 	char *header;
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *algo = repo->hash_algo;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
+	struct object_id compat_oid;
 	int hdrlen, status = 0;
+	int compat_type = -1;
+
+	if (compat) {
+		compat_type = type_from_string_gently(type, -1, 1);
+		if (compat_type == OBJ_BLOB)
+			hash_object_file(compat, buf, len, compat_type,
+					 &compat_oid);
+		else if (compat_type != -1) {
+			struct strbuf converted = STRBUF_INIT;
+			convert_object_file(&converted, algo, compat,
+					    buf, len, compat_type, 0);
+			hash_object_file(compat, converted.buf, converted.len,
+					 compat_type, &compat_oid);
+			strbuf_release(&converted);
+		}
+	}
 
 	/* type string, SP, %lu of the length plus NUL must fit this */
 	hdrlen = strlen(type) + MAX_HEADER_LEN;
@@ -2244,6 +2309,8 @@ int write_object_file_literally(const void *buf, unsigned long len,
 	if (freshen_packed_object(oid) || freshen_loose_object(oid))
 		goto cleanup;
 	status = write_loose_object(oid, header, hdrlen, buf, len, 0, 0);
+	if (compat_type != -1)
+		return repo_add_loose_object_map(repo, oid, &compat_oid);
 
 cleanup:
 	free(header);
@@ -2252,9 +2319,12 @@ int write_object_file_literally(const void *buf, unsigned long len,
 
 int force_object_loose(const struct object_id *oid, time_t mtime)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
 	void *buf;
 	unsigned long len;
 	struct object_info oi = OBJECT_INFO_INIT;
+	struct object_id compat_oid;
 	enum object_type type;
 	char hdr[MAX_HEADER_LEN];
 	int hdrlen;
@@ -2267,8 +2337,15 @@ int force_object_loose(const struct object_id *oid, time_t mtime)
 	oi.contentp = &buf;
 	if (oid_object_info_extended(the_repository, oid, &oi, 0))
 		return error(_("cannot read object for %s"), oid_to_hex(oid));
+	if (compat) {
+		if (repo_oid_to_algop(repo, oid, compat, &compat_oid))
+			return error(_("cannot map object %s to %s"),
+				     oid_to_hex(oid), compat->name);
+	}
 	hdrlen = format_object_header(hdr, sizeof(hdr), type, len);
 	ret = write_loose_object(oid, hdr, hdrlen, buf, len, mtime, 0);
+	if (!ret && compat)
+		ret = repo_add_loose_object_map(the_repository, oid, &compat_oid);
 	free(buf);
 
 	return ret;
-- 
2.41.0
Previous: Patrick SteinhardtNext: Patrick Steinhardt
Message 74 of 104 in “Initial support for multiple hash functions”
  1. 00/30 Initial support for multiple hash functionsEric W. Biederman, Sep 27, 2023
  2. 01/30 object-file-convert: Stubs for converting from one object format to anotherEric W. Biederman, Sep 27, 2023
  3. Eric SunshineSep 27, 2023
  4. Eric W. BiedermanOct 2, 2023
  5. Eric SunshineOct 2, 2023
  6. 02/30 oid-array: Teach oid-array to handle multiple kinds of oidsEric W. Biederman, Sep 27, 2023
  7. Eric SunshineSep 27, 2023
  8. 04/30 repository: add a compatibility hash algorithmEric W. Biederman, Sep 27, 2023
  9. 03/30 object-names: Support input of oids in any supported hashEric W. Biederman, Sep 27, 2023
  10. Eric SunshineSep 27, 2023
  11. Eric W. BiedermanOct 2, 2023
  12. 05/30 loose: add a mapping between SHA-1 and SHA-256 for loose objectsEric W. Biederman, Sep 27, 2023
  13. Eric SunshineSep 28, 2023
  14. Eric W. BiedermanOct 2, 2023
  15. Eric SunshineOct 2, 2023
  16. 06/30 loose: Compatibilty short name supportEric W. Biederman, Sep 27, 2023
  17. 08/30 object-file: Add a compat_oid_in parameter to write_object_file_flagsEric W. Biederman, Sep 27, 2023
  18. 07/30 object-file: Update the loose object map when writing loose objectsEric W. Biederman, Sep 27, 2023
  19. 09/30 commit: write commits for both hashesEric W. Biederman, Sep 27, 2023
  20. 10/30 commit: Convert mergetag before computing the signature of a commitEric W. Biederman, Sep 27, 2023
  21. 11/30 commit: Export add_header_signature to support handling signatures on tagsEric W. Biederman, Sep 27, 2023
  22. 12/30 tag: sign both hashesEric W. Biederman, Sep 27, 2023
  23. 14/30 object: Factor out parse_mode out of fast-import and tree-walk into in object.hEric W. Biederman, Sep 27, 2023
  24. 13/30 cache: add a function to read an OID of a specific algorithmEric W. Biederman, Sep 27, 2023
  25. 15/30 object-file-convert: add a function to convert trees between algorithmsEric W. Biederman, Sep 27, 2023
  26. 16/30 object-file-convert: convert tag objects when writingEric W. Biederman, Sep 27, 2023
  27. 17/30 object-file-convert: Don't leak when converting tag objectsEric W. Biederman, Sep 27, 2023
  28. 18/30 object-file-convert: convert commit objects when writingEric W. Biederman, Sep 27, 2023
  29. 19/30 object-file-convert: Convert commits that embed signed tagsEric W. Biederman, Sep 27, 2023
  30. 20/30 object-file: Update object_info_extended to reencode objectsEric W. Biederman, Sep 27, 2023
  31. 22/30 rev-parse: Add an --output-object-format parameterEric W. Biederman, Sep 27, 2023
  32. 21/30 repository: Implement extensions.compatObjectFormatEric W. Biederman, Sep 27, 2023
  33. Junio C HamanoSep 27, 2023
  34. Junio C HamanoSep 28, 2023
  35. Eric BiedermanSep 29, 2023
  36. Eric W. BiedermanSep 29, 2023
  37. Junio C HamanoSep 29, 2023
  38. Eric W. BiedermanOct 2, 2023
  39. Eric W. BiedermanOct 2, 2023
  40. 23/30 builtin/cat-file: Let the oid determine the output algorithmEric W. Biederman, Sep 27, 2023
  41. 25/30 object-file: Handle compat objects in check_object_signatureEric W. Biederman, Sep 27, 2023
  42. 26/30 builtin/ls-tree: Let the oid determine the output algorithmEric W. Biederman, Sep 27, 2023
  43. 24/30 tree-walk: init_tree_desc take an oid to get the hash algorithmEric W. Biederman, Sep 27, 2023
  44. 27/30 test-lib: Compute the compatibility hash so tests may use itEric W. Biederman, Sep 27, 2023
  45. 29/30 t1006: Test oid compatibility with cat-fileEric W. Biederman, Sep 27, 2023
  46. 28/30 t1006: Rename sha1 to oidEric W. Biederman, Sep 27, 2023
  47. 30/30 t1016-compatObjectFormat: Add tests to verify the conversion between objectsEric W. Biederman, Sep 27, 2023
  48. Junio C HamanoSep 27, 2023
  49. 00/30 initial support for multiple hash functionsEric W. Biederman, Oct 2, 2023
  50. 01/30 object-file-convert: stubs for converting from one object format to anotherEric W. Biederman, Oct 2, 2023
  51. Linus ArverFeb 8, 2024
  52. Patrick SteinhardtFeb 15, 2024
  53. 02/30 oid-array: teach oid-array to handle multiple kinds of oidsEric W. Biederman, Oct 2, 2023
  54. Linus ArverFeb 13, 2024
  55. Eric W. BiedermanFeb 15, 2024
  56. Linus ArverFeb 16, 2024
  57. Eric W. BiedermanFeb 16, 2024
  58. Linus ArverFeb 17, 2024
  59. Kristoffer HaugsbakkFeb 13, 2024
  60. Eric W. BiedermanFeb 15, 2024
  61. Patrick SteinhardtFeb 15, 2024
  62. 03/30 object-names: support input of oids in any supported hashEric W. Biederman, Oct 2, 2023
  63. Linus ArverFeb 13, 2024
  64. Patrick SteinhardtFeb 15, 2024
  65. 04/30 repository: add a compatibility hash algorithmEric W. Biederman, Oct 2, 2023
  66. Linus ArverFeb 13, 2024
  67. Patrick SteinhardtFeb 15, 2024
  68. 06/30 loose: compatibilty short name supportEric W. Biederman, Oct 2, 2023
  69. Patrick SteinhardtFeb 15, 2024
  70. 05/30 loose: add a mapping between SHA-1 and SHA-256 for loose objectsEric W. Biederman, Oct 2, 2023
  71. Linus ArverFeb 14, 2024
  72. Eric W. BiedermanFeb 15, 2024
  73. Patrick SteinhardtFeb 15, 2024
  74. 07/30 object-file: update the loose object map when writing loose objectsEric W. Biederman, Oct 2, 2023
  75. Patrick SteinhardtFeb 15, 2024
  76. 08/30 object-file: add a compat_oid_in parameter to write_object_file_flagsEric W. Biederman, Oct 2, 2023
  77. 10/30 commit: convert mergetag before computing the signature of a commitEric W. Biederman, Oct 2, 2023
  78. 09/30 commit: write commits for both hashesEric W. Biederman, Oct 2, 2023
  79. 11/30 commit: export add_header_signature to support handling signatures on tagsEric W. Biederman, Oct 2, 2023
  80. 12/30 tag: sign both hashesEric W. Biederman, Oct 2, 2023
  81. 13/30 cache: add a function to read an OID of a specific algorithmEric W. Biederman, Oct 2, 2023
  82. 14/30 object: factor out parse_mode out of fast-import and tree-walk into in object.hEric W. Biederman, Oct 2, 2023
  83. 15/30 object-file-convert: add a function to convert trees between algorithmsEric W. Biederman, Oct 2, 2023
  84. 16/30 object-file-convert: convert tag objects when writingEric W. Biederman, Oct 2, 2023
  85. 17/30 object-file-convert: don't leak when converting tag objectsEric W. Biederman, Oct 2, 2023
  86. 19/30 object-file-convert: convert commits that embed signed tagsEric W. Biederman, Oct 2, 2023
  87. 18/30 object-file-convert: convert commit objects when writingEric W. Biederman, Oct 2, 2023
  88. 20/30 object-file: update object_info_extended to reencode objectsEric W. Biederman, Oct 2, 2023
  89. 21/30 repository: implement extensions.compatObjectFormatEric W. Biederman, Oct 2, 2023
  90. 22/30 rev-parse: add an --output-object-format parameterEric W. Biederman, Oct 2, 2023
  91. Jean-Noël AvilaFeb 8, 2024
  92. 23/30 builtin/cat-file: let the oid determine the output algorithmEric W. Biederman, Oct 2, 2023
  93. 25/30 object-file: handle compat objects in check_object_signatureEric W. Biederman, Oct 2, 2023
  94. 26/30 builtin/ls-tree: let the oid determine the output algorithmEric W. Biederman, Oct 2, 2023
  95. 24/30 tree-walk: init_tree_desc take an oid to get the hash algorithmEric W. Biederman, Oct 2, 2023
  96. 27/30 test-lib: compute the compatibility hash so tests may use itEric W. Biederman, Oct 2, 2023
  97. 29/30 t1006: test oid compatibility with cat-fileEric W. Biederman, Oct 2, 2023
  98. 28/30 t1006: rename sha1 to oidEric W. Biederman, Oct 2, 2023
  99. 30/30 t1016-compatObjectFormat: add tests to verify the conversion between objectsEric W. Biederman, Oct 2, 2023
  100. Junio C HamanoFeb 7, 2024
  101. Linus ArverFeb 8, 2024
  102. Patrick SteinhardtFeb 8, 2024
  103. Linus ArverFeb 14, 2024
  104. Patrick SteinhardtFeb 15, 2024

Read the whole thread, see it on lore, or plain text.

$ cat FOOTERMessages come from the public archive at lore.kernel.org/git, fetched every hour. The front page is chosen and written each morning by an AI editor and can be wrong; the threads themselves are the record. About and API. For agents: an MCP server at https://gitlist.dev/mcp, and any thread, story or person page as Markdown by adding .md to its URL (or sending Accept: text/markdown). Details in /llms.txt.