git/list[1] front-page[2] threads[3] people[4] search[5] about
 

[PATCH 03/30] object-names: Support input of oids in any supported hash

From
EBEric W. Biederman <ebiederm@gmail.com>
Date
Sep 27, 2023, 19:55 UTC
Message-ID
<20230927195537.1682-3-ebiederm@gmail.com>
In-Reply-To
<87jzsbjt0a.fsf@gmail.froward.int.ebiederm.org>
From: "Eric W. Biederman" <ebiederm@xmission.com>

Support short oids encoded in any algorithm, while ensuring enough of the oid is specified to disambiguate between all of the oids in the repository encoded in any algorithm.

By default have the code continue to only accept oids specified in the storage hash algorithm of the repository, but when something is ambiguous display all of the possible oids from any oid encoding.

A new flag is added GET_OID_HASH_ANY that when supplied causes the code to accept oids specified in any hash algorithm, and to return the oids that were resolved.

This implements the functionality that allows both SHA-1 and SHA-256 object names, from the "Object names on the command line" section of the hash function transition document.

Care is taken in get_short_oid so that when the result is ambiguous the output remains the same of GIT_OID_HASH_ANY was not supplied. If GET_OID_HASH_ANY was supplied objects of any hash algorithm that match the prefix are displayed.

This required updating repo_for_each_abbrev to give it a parameter so that it knows to look at all hash algorithms.

Signed-off-by: "Eric W. Biederman" <ebiederm@xmission.com>
---
 builtin/rev-parse.c |  2 +-
 hash-ll.h           |  1 +
 object-name.c       | 49 +++++++++++++++++++++++++++++++++++----------
 object-name.h       |  3 ++-
 4 files changed, 42 insertions(+), 13 deletions(-)
diff --git a/builtin/rev-parse.c b/builtin/rev-parse.c
index fde8861ca4e0..43e96765400c 100644
--- a/builtin/rev-parse.c
+++ b/builtin/rev-parse.c
@@ -882,7 +882,7 @@ int cmd_rev_parse(int argc, const char **argv, const char *prefix)
 				continue;
 			}
 			if (skip_prefix(arg, "--disambiguate=", &arg)) {
-				repo_for_each_abbrev(the_repository, arg,
+				repo_for_each_abbrev(the_repository, arg, the_hash_algo,
 						     show_abbrev, NULL);
 				continue;
 			}
diff --git a/hash-ll.h b/hash-ll.h
index 10d84cc20888..2cfde63ae1cf 100644
--- a/hash-ll.h
+++ b/hash-ll.h
@@ -145,6 +145,7 @@ struct object_id {
 #define GET_OID_RECORD_PATH     0200
 #define GET_OID_ONLY_TO_DIE    04000
 #define GET_OID_REQUIRE_PATH  010000
+#define GET_OID_HASH_ANY      020000
 
 #define GET_OID_DISAMBIGUATORS \
 	(GET_OID_COMMIT | GET_OID_COMMITTISH | \
diff --git a/object-name.c b/object-name.c
index 0bfa29dbbfe9..976b7106821b 100644
--- a/object-name.c
+++ b/object-name.c
@@ -25,6 +25,7 @@
 #include "midx.h"
 #include "commit-reach.h"
 #include "date.h"
+#include "object-file-convert.h"
 
 static int get_oid_oneline(struct repository *r, const char *, struct object_id *, struct commit_list *);
 
@@ -49,6 +50,7 @@ struct disambiguate_state {
 
 static void update_candidates(struct disambiguate_state *ds, const struct object_id *current)
 {
+	/* The hash algorithm of the current has already been filtered */
 	if (ds->always_call_fn) {
 		ds->ambiguous = ds->fn(ds->repo, current, ds->cb_data) ? 1 : 0;
 		return;
@@ -134,6 +136,8 @@ static void unique_in_midx(struct multi_pack_index *m,
 {
 	uint32_t num, i, first = 0;
 	const struct object_id *current = NULL;
+	int len = ds->len > ds->repo->hash_algo->hexsz ?
+		ds->repo->hash_algo->hexsz : ds->len;
 	num = m->num_objects;
 
 	if (!num)
@@ -149,7 +153,7 @@ static void unique_in_midx(struct multi_pack_index *m,
 	for (i = first; i < num && !ds->ambiguous; i++) {
 		struct object_id oid;
 		current = nth_midxed_object_oid(&oid, m, i);
-		if (!match_hash(ds->len, ds->bin_pfx.hash, current->hash))
+		if (!match_hash(len, ds->bin_pfx.hash, current->hash))
 			break;
 		update_candidates(ds, current);
 	}
@@ -159,6 +163,8 @@ static void unique_in_pack(struct packed_git *p,
 			   struct disambiguate_state *ds)
 {
 	uint32_t num, i, first = 0;
+	int len = ds->len > ds->repo->hash_algo->hexsz ?
+		ds->repo->hash_algo->hexsz : ds->len;
 
 	if (p->multi_pack_index)
 		return;
@@ -177,7 +183,7 @@ static void unique_in_pack(struct packed_git *p,
 	for (i = first; i < num && !ds->ambiguous; i++) {
 		struct object_id oid;
 		nth_packed_object_id(&oid, p, i);
-		if (!match_hash(ds->len, ds->bin_pfx.hash, oid.hash))
+		if (!match_hash(len, ds->bin_pfx.hash, oid.hash))
 			break;
 		update_candidates(ds, &oid);
 	}
@@ -188,6 +194,10 @@ static void find_short_packed_object(struct disambiguate_state *ds)
 	struct multi_pack_index *m;
 	struct packed_git *p;
 
+	/* Skip, unless oids from the storage hash algorithm are wanted */
+	if (ds->bin_pfx.algo && (&hash_algos[ds->bin_pfx.algo] != ds->repo->hash_algo))
+		return;
+
 	for (m = get_multi_pack_index(ds->repo); m && !ds->ambiguous;
 	     m = m->next)
 		unique_in_midx(m, ds);
@@ -326,11 +336,12 @@ int set_disambiguate_hint_config(const char *var, const char *value)
 
 static int init_object_disambiguation(struct repository *r,
 				      const char *name, int len,
+				      const struct git_hash_algo *algo,
 				      struct disambiguate_state *ds)
 {
 	int i;
 
-	if (len < MINIMUM_ABBREV || len > the_hash_algo->hexsz)
+	if (len < MINIMUM_ABBREV || len > GIT_MAX_HEXSZ)
 		return -1;
 
 	memset(ds, 0, sizeof(*ds));
@@ -357,6 +368,7 @@ static int init_object_disambiguation(struct repository *r,
 	ds->len = len;
 	ds->hex_pfx[len] = '\0';
 	ds->repo = r;
+	ds->bin_pfx.algo = algo ? hash_algo_by_ptr(algo) : GIT_HASH_UNKNOWN;
 	prepare_alt_odb(r);
 	return 0;
 }
@@ -491,9 +503,10 @@ static int repo_collect_ambiguous(struct repository *r UNUSED,
 	return collect_ambiguous(oid, data);
 }
 
-static int sort_ambiguous(const void *a, const void *b, void *ctx)
+static int sort_ambiguous(const void *va, const void *vb, void *ctx)
 {
 	struct repository *sort_ambiguous_repo = ctx;
+	const struct object_id *a = va, *b = vb;
 	int a_type = oid_object_info(sort_ambiguous_repo, a, NULL);
 	int b_type = oid_object_info(sort_ambiguous_repo, b, NULL);
 	int a_type_sort;
@@ -503,8 +516,13 @@ static int sort_ambiguous(const void *a, const void *b, void *ctx)
 	 * Sorts by hash within the same object type, just as
 	 * oid_array_for_each_unique() would do.
 	 */
-	if (a_type == b_type)
-		return oidcmp(a, b);
+	if (a_type == b_type) {
+		/* Is the hash algorithm the same? */
+		if (a->algo == b->algo)
+			return oidcmp(a, b);
+		else
+			return a->algo > b->algo ? 1 : -1;
+	}
 
 	/*
 	 * Between object types show tags, then commits, and finally
@@ -533,8 +551,12 @@ static enum get_oid_result get_short_oid(struct repository *r,
 	int status;
 	struct disambiguate_state ds;
 	int quietly = !!(flags & GET_OID_QUIETLY);
+	const struct git_hash_algo *algo = r->hash_algo;
+
+	if (flags & GET_OID_HASH_ANY)
+		algo = NULL;
 
-	if (init_object_disambiguation(r, name, len, &ds) < 0)
+	if (init_object_disambiguation(r, name, len, algo, &ds) < 0)
 		return -1;
 
 	if (HAS_MULTI_BITS(flags & GET_OID_DISAMBIGUATORS))
@@ -553,6 +575,7 @@ static enum get_oid_result get_short_oid(struct repository *r,
 	else
 		ds.fn = default_disambiguate_hint;
 
+
 	find_short_object_filename(&ds);
 	find_short_packed_object(&ds);
 	status = finish_object_disambiguation(&ds, oid);
@@ -588,7 +611,7 @@ static enum get_oid_result get_short_oid(struct repository *r,
 		if (!ds.ambiguous)
 			ds.fn = NULL;
 
-		repo_for_each_abbrev(r, ds.hex_pfx, collect_ambiguous, &collect);
+		repo_for_each_abbrev(r, ds.hex_pfx, algo, collect_ambiguous, &collect);
 		sort_ambiguous_oid_array(r, &collect);
 
 		if (oid_array_for_each(&collect, show_ambiguous_object, &out))
@@ -610,15 +633,17 @@ static enum get_oid_result get_short_oid(struct repository *r,
 }
 
 int repo_for_each_abbrev(struct repository *r, const char *prefix,
+			 const struct git_hash_algo *algo,
 			 each_abbrev_fn fn, void *cb_data)
 {
 	struct oid_array collect = OID_ARRAY_INIT;
 	struct disambiguate_state ds;
 	int ret;
 
-	if (init_object_disambiguation(r, prefix, strlen(prefix), &ds) < 0)
+	if (init_object_disambiguation(r, prefix, strlen(prefix), algo, &ds) < 0)
 		return -1;
 
+	ds.bin_pfx.algo = GIT_HASH_UNKNOWN;
 	ds.always_call_fn = 1;
 	ds.fn = repo_collect_ambiguous;
 	ds.cb_data = &collect;
@@ -787,10 +812,12 @@ void strbuf_add_unique_abbrev(struct strbuf *sb, const struct object_id *oid,
 int repo_find_unique_abbrev_r(struct repository *r, char *hex,
 			      const struct object_id *oid, int len)
 {
+	const struct git_hash_algo *algo =
+		oid->algo ? &hash_algos[oid->algo] : r->hash_algo;
 	struct disambiguate_state ds;
 	struct min_abbrev_data mad;
 	struct object_id oid_ret;
-	const unsigned hexsz = r->hash_algo->hexsz;
+	const unsigned hexsz = algo->hexsz;
 
 	if (len < 0) {
 		unsigned long count = repo_approximate_object_count(r);
@@ -826,7 +853,7 @@ int repo_find_unique_abbrev_r(struct repository *r, char *hex,
 
 	find_abbrev_len_packed(&mad);
 
-	if (init_object_disambiguation(r, hex, mad.cur_len, &ds) < 0)
+	if (init_object_disambiguation(r, hex, mad.cur_len, algo, &ds) < 0)
 		return -1;
 
 	ds.fn = repo_extend_abbrev_len;
diff --git a/object-name.h b/object-name.h
index 9ae522307148..064ddc97d1fe 100644
--- a/object-name.h
+++ b/object-name.h
@@ -67,7 +67,8 @@ enum get_oid_result get_oid_with_context(struct repository *repo, const char *st
 
 
 typedef int each_abbrev_fn(const struct object_id *oid, void *);
-int repo_for_each_abbrev(struct repository *r, const char *prefix, each_abbrev_fn, void *);
+int repo_for_each_abbrev(struct repository *r, const char *prefix,
+			 const struct git_hash_algo *algo, each_abbrev_fn, void *);
 
 int set_disambiguate_hint_config(const char *var, const char *value);
 
-- 
2.41.0
Previous: Eric W. BiedermanNext: Eric Sunshine
Message 9 of 104 in “Initial support for multiple hash functions”
  1. 00/30 Initial support for multiple hash functionsEric W. Biederman, Sep 27, 2023
  2. 01/30 object-file-convert: Stubs for converting from one object format to anotherEric W. Biederman, Sep 27, 2023
  3. Eric SunshineSep 27, 2023
  4. Eric W. BiedermanOct 2, 2023
  5. Eric SunshineOct 2, 2023
  6. 02/30 oid-array: Teach oid-array to handle multiple kinds of oidsEric W. Biederman, Sep 27, 2023
  7. Eric SunshineSep 27, 2023
  8. 04/30 repository: add a compatibility hash algorithmEric W. Biederman, Sep 27, 2023
  9. 03/30 object-names: Support input of oids in any supported hashEric W. Biederman, Sep 27, 2023
  10. Eric SunshineSep 27, 2023
  11. Eric W. BiedermanOct 2, 2023
  12. 05/30 loose: add a mapping between SHA-1 and SHA-256 for loose objectsEric W. Biederman, Sep 27, 2023
  13. Eric SunshineSep 28, 2023
  14. Eric W. BiedermanOct 2, 2023
  15. Eric SunshineOct 2, 2023
  16. 06/30 loose: Compatibilty short name supportEric W. Biederman, Sep 27, 2023
  17. 08/30 object-file: Add a compat_oid_in parameter to write_object_file_flagsEric W. Biederman, Sep 27, 2023
  18. 07/30 object-file: Update the loose object map when writing loose objectsEric W. Biederman, Sep 27, 2023
  19. 09/30 commit: write commits for both hashesEric W. Biederman, Sep 27, 2023
  20. 10/30 commit: Convert mergetag before computing the signature of a commitEric W. Biederman, Sep 27, 2023
  21. 11/30 commit: Export add_header_signature to support handling signatures on tagsEric W. Biederman, Sep 27, 2023
  22. 12/30 tag: sign both hashesEric W. Biederman, Sep 27, 2023
  23. 14/30 object: Factor out parse_mode out of fast-import and tree-walk into in object.hEric W. Biederman, Sep 27, 2023
  24. 13/30 cache: add a function to read an OID of a specific algorithmEric W. Biederman, Sep 27, 2023
  25. 15/30 object-file-convert: add a function to convert trees between algorithmsEric W. Biederman, Sep 27, 2023
  26. 16/30 object-file-convert: convert tag objects when writingEric W. Biederman, Sep 27, 2023
  27. 17/30 object-file-convert: Don't leak when converting tag objectsEric W. Biederman, Sep 27, 2023
  28. 18/30 object-file-convert: convert commit objects when writingEric W. Biederman, Sep 27, 2023
  29. 19/30 object-file-convert: Convert commits that embed signed tagsEric W. Biederman, Sep 27, 2023
  30. 20/30 object-file: Update object_info_extended to reencode objectsEric W. Biederman, Sep 27, 2023
  31. 22/30 rev-parse: Add an --output-object-format parameterEric W. Biederman, Sep 27, 2023
  32. 21/30 repository: Implement extensions.compatObjectFormatEric W. Biederman, Sep 27, 2023
  33. Junio C HamanoSep 27, 2023
  34. Junio C HamanoSep 28, 2023
  35. Eric BiedermanSep 29, 2023
  36. Eric W. BiedermanSep 29, 2023
  37. Junio C HamanoSep 29, 2023
  38. Eric W. BiedermanOct 2, 2023
  39. Eric W. BiedermanOct 2, 2023
  40. 23/30 builtin/cat-file: Let the oid determine the output algorithmEric W. Biederman, Sep 27, 2023
  41. 25/30 object-file: Handle compat objects in check_object_signatureEric W. Biederman, Sep 27, 2023
  42. 26/30 builtin/ls-tree: Let the oid determine the output algorithmEric W. Biederman, Sep 27, 2023
  43. 24/30 tree-walk: init_tree_desc take an oid to get the hash algorithmEric W. Biederman, Sep 27, 2023
  44. 27/30 test-lib: Compute the compatibility hash so tests may use itEric W. Biederman, Sep 27, 2023
  45. 29/30 t1006: Test oid compatibility with cat-fileEric W. Biederman, Sep 27, 2023
  46. 28/30 t1006: Rename sha1 to oidEric W. Biederman, Sep 27, 2023
  47. 30/30 t1016-compatObjectFormat: Add tests to verify the conversion between objectsEric W. Biederman, Sep 27, 2023
  48. Junio C HamanoSep 27, 2023
  49. 00/30 initial support for multiple hash functionsEric W. Biederman, Oct 2, 2023
  50. 01/30 object-file-convert: stubs for converting from one object format to anotherEric W. Biederman, Oct 2, 2023
  51. Linus ArverFeb 8, 2024
  52. Patrick SteinhardtFeb 15, 2024
  53. 02/30 oid-array: teach oid-array to handle multiple kinds of oidsEric W. Biederman, Oct 2, 2023
  54. Linus ArverFeb 13, 2024
  55. Eric W. BiedermanFeb 15, 2024
  56. Linus ArverFeb 16, 2024
  57. Eric W. BiedermanFeb 16, 2024
  58. Linus ArverFeb 17, 2024
  59. Kristoffer HaugsbakkFeb 13, 2024
  60. Eric W. BiedermanFeb 15, 2024
  61. Patrick SteinhardtFeb 15, 2024
  62. 03/30 object-names: support input of oids in any supported hashEric W. Biederman, Oct 2, 2023
  63. Linus ArverFeb 13, 2024
  64. Patrick SteinhardtFeb 15, 2024
  65. 04/30 repository: add a compatibility hash algorithmEric W. Biederman, Oct 2, 2023
  66. Linus ArverFeb 13, 2024
  67. Patrick SteinhardtFeb 15, 2024
  68. 06/30 loose: compatibilty short name supportEric W. Biederman, Oct 2, 2023
  69. Patrick SteinhardtFeb 15, 2024
  70. 05/30 loose: add a mapping between SHA-1 and SHA-256 for loose objectsEric W. Biederman, Oct 2, 2023
  71. Linus ArverFeb 14, 2024
  72. Eric W. BiedermanFeb 15, 2024
  73. Patrick SteinhardtFeb 15, 2024
  74. 07/30 object-file: update the loose object map when writing loose objectsEric W. Biederman, Oct 2, 2023
  75. Patrick SteinhardtFeb 15, 2024
  76. 08/30 object-file: add a compat_oid_in parameter to write_object_file_flagsEric W. Biederman, Oct 2, 2023
  77. 10/30 commit: convert mergetag before computing the signature of a commitEric W. Biederman, Oct 2, 2023
  78. 09/30 commit: write commits for both hashesEric W. Biederman, Oct 2, 2023
  79. 11/30 commit: export add_header_signature to support handling signatures on tagsEric W. Biederman, Oct 2, 2023
  80. 12/30 tag: sign both hashesEric W. Biederman, Oct 2, 2023
  81. 13/30 cache: add a function to read an OID of a specific algorithmEric W. Biederman, Oct 2, 2023
  82. 14/30 object: factor out parse_mode out of fast-import and tree-walk into in object.hEric W. Biederman, Oct 2, 2023
  83. 15/30 object-file-convert: add a function to convert trees between algorithmsEric W. Biederman, Oct 2, 2023
  84. 16/30 object-file-convert: convert tag objects when writingEric W. Biederman, Oct 2, 2023
  85. 17/30 object-file-convert: don't leak when converting tag objectsEric W. Biederman, Oct 2, 2023
  86. 19/30 object-file-convert: convert commits that embed signed tagsEric W. Biederman, Oct 2, 2023
  87. 18/30 object-file-convert: convert commit objects when writingEric W. Biederman, Oct 2, 2023
  88. 20/30 object-file: update object_info_extended to reencode objectsEric W. Biederman, Oct 2, 2023
  89. 21/30 repository: implement extensions.compatObjectFormatEric W. Biederman, Oct 2, 2023
  90. 22/30 rev-parse: add an --output-object-format parameterEric W. Biederman, Oct 2, 2023
  91. Jean-Noël AvilaFeb 8, 2024
  92. 23/30 builtin/cat-file: let the oid determine the output algorithmEric W. Biederman, Oct 2, 2023
  93. 25/30 object-file: handle compat objects in check_object_signatureEric W. Biederman, Oct 2, 2023
  94. 26/30 builtin/ls-tree: let the oid determine the output algorithmEric W. Biederman, Oct 2, 2023
  95. 24/30 tree-walk: init_tree_desc take an oid to get the hash algorithmEric W. Biederman, Oct 2, 2023
  96. 27/30 test-lib: compute the compatibility hash so tests may use itEric W. Biederman, Oct 2, 2023
  97. 29/30 t1006: test oid compatibility with cat-fileEric W. Biederman, Oct 2, 2023
  98. 28/30 t1006: rename sha1 to oidEric W. Biederman, Oct 2, 2023
  99. 30/30 t1016-compatObjectFormat: add tests to verify the conversion between objectsEric W. Biederman, Oct 2, 2023
  100. Junio C HamanoFeb 7, 2024
  101. Linus ArverFeb 8, 2024
  102. Patrick SteinhardtFeb 8, 2024
  103. Linus ArverFeb 14, 2024
  104. Patrick SteinhardtFeb 15, 2024

Read the whole thread, see it on lore, or plain text.

$ cat FOOTERMessages come from the public archive at lore.kernel.org/git, fetched every hour. The front page is chosen and written each morning by an AI editor and can be wrong; the threads themselves are the record. About and API. For agents: an MCP server at https://gitlist.dev/mcp, and any thread, story or person page as Markdown by adding .md to its URL (or sending Accept: text/markdown). Details in /llms.txt.