git/list[1] front-page[2] threads[3] people[4] search[5] about
 

[PATCH 27/32] builtin/fast-import: compute compatibility hashs for imported objects

From
Eric W. Biederman <ebiederm@xmission.com>
Date
Sep 8, 2023, 23:10 UTC
Message-ID
<20230908231049.2035003-27-ebiederm@xmission.com>
In-Reply-To
<87sf7ol0z3.fsf@email.froward.int.ebiederm.org>

When the code is in dual hash mode for every object fast-import creates compute the standard oid and it's compatibility mapping. The compatibility mapping is stored in struct pack_idx_entry so that it can be used when an index is created.

For fast-import the code needs to be careful because when a new object only refers to other newly created objects the compatibility mapping for those new objects is not stored anywhere permanently. So have the code first look the the compatibility oid in the newly created objects, and then look for the compatibilty oid in the standard mapping tables.

As fast-import requires objects to be specified before the objects that reference them nothing special needs to happen to deal with out of order objects.

Signed-off-by: "Eric W. Biederman" <ebiederm@xmission.com>
---
 builtin/fast-import.c | 89 +++++++++++++++++++++++++++++++++++++------
 1 file changed, 77 insertions(+), 12 deletions(-)
diff --git a/builtin/fast-import.c b/builtin/fast-import.c
index 2c645fcfbe3f..f1c250dd3c8f 100644
--- a/builtin/fast-import.c
+++ b/builtin/fast-import.c
@@ -26,6 +26,8 @@
 #include "commit-reach.h"
 #include "khash.h"
 #include "date.h"
+#include "object-file-convert.h"
+#include "pack-compat-map.h"
 
 #define PACK_ID_BITS 16
 #define MAX_PACK_ID ((1<<PACK_ID_BITS)-1)
@@ -775,9 +777,14 @@ static void start_packfile(void)
 	all_packs[pack_id] = p;
 }
 
-static const char *create_index(void)
+struct pack_index_names {
+	const char *index_name;
+	const char *compat_name;
+};
+
+static struct pack_index_names create_index(void)
 {
-	const char *tmpfile;
+	struct pack_index_names tmp = {};
 	struct pack_idx_entry **idx, **c, **last;
 	struct object_entry *e;
 	struct object_entry_pool *o;
@@ -793,13 +800,15 @@ static const char *create_index(void)
 	if (c != last)
 		die("internal consistency error creating the index");
 
-	tmpfile = write_idx_file(NULL, idx, object_count, &pack_idx_opts,
-				 pack_data->hash);
+	tmp.index_name = write_idx_file(NULL, idx, object_count, &pack_idx_opts,
+					pack_data->hash);
+	tmp.compat_name = write_compat_map_file(NULL, idx, object_count,
+						pack_data->hash);
 	free(idx);
-	return tmpfile;
+	return tmp;
 }
 
-static char *keep_pack(const char *curr_index_name)
+static char *keep_pack(struct pack_index_names curr)
 {
 	static const char *keep_msg = "fast-import";
 	struct strbuf name = STRBUF_INIT;
@@ -818,9 +827,17 @@ static char *keep_pack(const char *curr_index_name)
 		die("cannot store pack file");
 
 	odb_pack_name(&name, pack_data->hash, "idx");
-	if (finalize_object_file(curr_index_name, name.buf))
+	if (finalize_object_file(curr.index_name, name.buf))
 		die("cannot store index file");
-	free((void *)curr_index_name);
+
+	if (curr.compat_name) {
+		odb_pack_name(&name, pack_data->hash, "compat");
+		if (finalize_object_file(curr.compat_name, name.buf))
+			die("cannot store compatibility map file");
+	}
+
+	free((void *)curr.index_name);
+	free((void *)curr.compat_name);
 	return strbuf_detach(&name, NULL);
 }
 
@@ -943,6 +960,8 @@ static int store_object(
 	struct object_id *oidout,
 	uintmax_t mark)
 {
+	struct repository *repo = the_repository;
+	const struct git_hash_algo *compat = repo->compat_hash_algo;
 	void *out, *delta;
 	struct object_entry *e;
 	unsigned char hdr[96];
@@ -966,8 +985,7 @@ static int store_object(
 	if (e->idx.offset) {
 		duplicate_count_by_type[type]++;
 		return 1;
-	} else if (find_sha1_pack(oid.hash,
-				  get_all_packs(the_repository))) {
+	} else if (find_sha1_pack(oid.hash, get_all_packs(repo))) {
 		e->type = type;
 		e->pack_id = MAX_PACK_ID;
 		e->idx.offset = 1; /* just not zero! */
@@ -1026,6 +1044,42 @@ static int store_object(
 	e->type = type;
 	e->pack_id = pack_id;
 	e->idx.offset = pack_size;
+	if (compat && (type == OBJ_BLOB)) {
+		compat->init_fn(&c);
+		compat->update_fn(&c, hdr, hdrlen);
+		compat->update_fn(&c, dat->buf, dat->len);
+		compat->final_oid_fn(&e->idx.compat_oid, &c);
+	} else if (compat) {
+		struct object_file_convert_state state;
+		struct strbuf out = STRBUF_INIT;
+		int ret;
+
+		convert_object_file_begin(&state, &out, the_hash_algo, compat,
+					  dat->buf, dat->len, type);
+		for (;;) {
+			struct object_entry *pobj;
+
+			convert_object_file_step(&state);
+			if (ret != 1)
+				break;
+
+			ret = -1;
+			pobj = find_object(&state.oid);
+			if (pobj && pobj->idx.compat_oid.algo)
+				oidcpy(&state.mapped_oid, &pobj->idx.compat_oid);
+			else if (pobj)
+				break;
+			else if (repo_oid_to_algop(repo, &state.oid, compat,
+						   &state.mapped_oid))
+				break;
+		}
+		convert_object_file_end(&state, ret);
+		if (ret)
+			die(_("No mapping for %s to %s\n"),
+			    oid_to_hex(&state.oid), compat->name);
+		hash_object_file(compat, out.buf, out.len, type, &e->idx.compat_oid);
+		strbuf_release(&out);
+	}
 	object_count++;
 	object_count_by_type[type]++;
 
@@ -1084,14 +1138,15 @@ static void truncate_pack(struct hashfile_checkpoint *checkpoint)
 
 static void stream_blob(uintmax_t len, struct object_id *oidout, uintmax_t mark)
 {
+	const struct git_hash_algo *compat = the_repository->compat_hash_algo;
 	size_t in_sz = 64 * 1024, out_sz = 64 * 1024;
 	unsigned char *in_buf = xmalloc(in_sz);
 	unsigned char *out_buf = xmalloc(out_sz);
 	struct object_entry *e;
-	struct object_id oid;
+	struct object_id oid, compat_oid;
 	unsigned long hdrlen;
 	off_t offset;
-	git_hash_ctx c;
+	git_hash_ctx c, compat_c;
 	git_zstream s;
 	struct hashfile_checkpoint checkpoint;
 	int status = Z_OK;
@@ -1109,6 +1164,10 @@ static void stream_blob(uintmax_t len, struct object_id *oidout, uintmax_t mark)
 
 	the_hash_algo->init_fn(&c);
 	the_hash_algo->update_fn(&c, out_buf, hdrlen);
+	if (compat) {
+		compat->init_fn(&compat_c);
+		compat->update_fn(&compat_c, out_buf, hdrlen);
+	}
 
 	crc32_begin(pack_file);
 
@@ -1127,6 +1186,8 @@ static void stream_blob(uintmax_t len, struct object_id *oidout, uintmax_t mark)
 				die("EOF in data (%" PRIuMAX " bytes remaining)", len);
 
 			the_hash_algo->update_fn(&c, in_buf, n);
+			if (compat)
+				compat->update_fn(&compat_c, in_buf, n);
 			s.next_in = in_buf;
 			s.avail_in = n;
 			len -= n;
@@ -1153,6 +1214,8 @@ static void stream_blob(uintmax_t len, struct object_id *oidout, uintmax_t mark)
 	}
 	git_deflate_end(&s);
 	the_hash_algo->final_oid_fn(&oid, &c);
+	if (compat)
+		compat->final_oid_fn(&compat_oid, &compat_c);
 
 	if (oidout)
 		oidcpy(oidout, &oid);
@@ -1180,6 +1243,8 @@ static void stream_blob(uintmax_t len, struct object_id *oidout, uintmax_t mark)
 		e->pack_id = pack_id;
 		e->idx.offset = offset;
 		e->idx.crc32 = crc32_end(pack_file);
+		if (compat)
+			oidcpy(&e->idx.compat_oid, &compat_oid);
 		object_count++;
 		object_count_by_type[OBJ_BLOB]++;
 	}
-- 
2.41.0
Previous: Junio C HamanoNext: Eric W. Biederman
Message 19 of 59 in “SHA256 and SHA1 interoperability”
  1. Eric W. BiedermanSep 8, 2023
  2. 02/32 doc hash-function-transition: Replace compatObjectFormat with compatMapEric W. Biederman, Sep 8, 2023
  3. brian m. carlsonSep 10, 2023
  4. Eric W. BiedermanSep 10, 2023
  5. Junio C HamanoSep 11, 2023
  6. 02/32 doc hash-function-transition: Replace compatObjectFormat with mapObjectFormatEric W. Biederman, Sep 11, 2023
  7. 02/32 doc hash-function-transition: Augment compatObjectFormat with readCompatMapEric W. Biederman, Sep 11, 2023
  8. Oswald BuddenhagenSep 12, 2023
  9. Eric W. BiedermanSep 12, 2023
  10. Oswald BuddenhagenSep 13, 2023
  11. 04/32 object-name: Initial support for ^{sha1} and ^{sha256}Eric W. Biederman, Sep 8, 2023
  12. 06/32 repository: Implement core.compatMapEric W. Biederman, Sep 8, 2023
  13. 07/32 loose: add a mapping between SHA-1 and SHA-256 for loose objectsEric W. Biederman, Sep 8, 2023
  14. 19/32 object-file-convert: convert tag commits when writingEric W. Biederman, Sep 8, 2023
  15. 20/32 builtin/cat-file: Let the oid determine the output algorithmEric W. Biederman, Sep 8, 2023
  16. 22/32 object-file: Handle compat objects in check_object_signatureEric W. Biederman, Sep 8, 2023
  17. 26/32 object-file-convert: Implement convert_object_file_{begin,step,end}Eric W. Biederman, Sep 8, 2023
  18. Junio C HamanoSep 11, 2023
  19. 27/32 builtin/fast-import: compute compatibility hashs for imported objectsEric W. Biederman, Sep 8, 2023
  20. 29/32 builtin/index-pack: Compute the compatibility hashEric W. Biederman, Sep 8, 2023
  21. 31/32 unpack-objects: Update to compute and write the compatibility hashesEric W. Biederman, Sep 8, 2023
  22. 16/32 object: Factor out parse_mode out of fast-import and tree-walk into in object.hEric W. Biederman, Sep 8, 2023
  23. 10/32 bulk-checkin: Only accept blobsEric W. Biederman, Sep 8, 2023
  24. 23/32 builtin/ls-tree: Let the oid determine the output algorithmEric W. Biederman, Sep 8, 2023
  25. 12/32 bulk-checkin: hash object with compatibility algorithmEric W. Biederman, Sep 8, 2023
  26. Junio C HamanoSep 11, 2023
  27. 14/32 commit: write commits for both hashesEric W. Biederman, Sep 8, 2023
  28. Junio C HamanoSep 11, 2023
  29. 03/32 object-file-convert: Stubs for converting from one object format to anotherEric W. Biederman, Sep 8, 2023
  30. 08/32 loose: Compatibilty short name supportEric W. Biederman, Sep 8, 2023
  31. 01/32 doc hash-file-transition: A map file for mapping between sha1 and sha256Eric W. Biederman, Sep 8, 2023
  32. brian m. carlsonSep 10, 2023
  33. Eric W. BiedermanSep 10, 2023
  34. brian m. carlsonSep 12, 2023
  35. Eric W. BiedermanSep 12, 2023
  36. 15/32 cache: add a function to read an OID of a specific algorithmEric W. Biederman, Sep 8, 2023
  37. 32/32 object-file-convert: Implement repo_submodule_oid_to_algopEric W. Biederman, Sep 8, 2023
  38. 30/32 builtin/index-pack: Make the stack in compute_compat_oid explicitEric W. Biederman, Sep 8, 2023
  39. 28/32 builtin/index-pack: Add a simple oid indexEric W. Biederman, Sep 8, 2023
  40. 25/32 pack-compat-map: Add support for .compat files of a packfileEric W. Biederman, Sep 8, 2023
  41. Junio C HamanoSep 11, 2023
  42. Taylor BlauOct 5, 2023
  43. 21/32 tree-walk: init_tree_desc take an oid to get the hash algorithmEric W. Biederman, Sep 8, 2023
  44. 24/32 builtin/pack-objects: Communicate the compatibility hash through struct pack_idx_entryEric W. Biederman, Sep 8, 2023
  45. 18/32 object-file-convert: convert commit objects when writingEric W. Biederman, Sep 8, 2023
  46. 17/32 object-file-convert: add a function to convert trees between algorithmsEric W. Biederman, Sep 8, 2023
  47. 09/32 object-file: Update the loose object map when writing loose objectsEric W. Biederman, Sep 8, 2023
  48. 11/32 pack: Communicate the compat_oid through struct pack_idx_entryEric W. Biederman, Sep 8, 2023
  49. 05/32 repository: add a compatibility hash algorithmEric W. Biederman, Sep 8, 2023
  50. 13/32 object-file: Add a compat_oid_in parameter to write_object_file_flagsEric W. Biederman, Sep 8, 2023
  51. Eric W. BiedermanSep 9, 2023
  52. brian m. carlsonSep 10, 2023
  53. Eric W. BiedermanSep 10, 2023
  54. Junio C HamanoSep 11, 2023
  55. Eric W. BiedermanSep 11, 2023
  56. brian m. carlsonSep 11, 2023
  57. Eric W. BiedermanSep 12, 2023
  58. Junio C HamanoSep 12, 2023
  59. Eric W. BiedermanSep 14, 2023

Read the whole thread, see it on lore, or plain text.

$ cat FOOTERMessages come from the public archive at lore.kernel.org/git, fetched every hour. The front page is chosen and written each morning by an AI editor and can be wrong; the threads themselves are the record. About and API. For agents: an MCP server at https://gitlist.dev/mcp, and any thread, story or person page as Markdown by adding .md to its URL (or sending Accept: text/markdown). Details in /llms.txt.