{"thread":{"id":"54","subject":"[PATCH] Get commits from remote repositories by HTTP","startedAt":"2005-04-16T22:03:51Z","lastAt":"2005-04-18T20:48:50Z","messageCount":14,"participants":["Daniel Barkalow","Martin Mares","Tony Luck","Jan-Benedict Glaw","Adam Kropelin","tony.luck@intel.com","Petr Baudis"],"isPatch":true,"patchVersion":1,"patchTotal":null},"messages":[{"id":"357","messageId":"Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org","threadId":"54","inReplyTo":null,"subject":"[PATCH] Get commits from remote repositories by HTTP","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-04-16T22:03:51Z","receivedAt":"2005-04-16T22:03:51Z","isPatch":true,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"This adds a program to download a commit, the trees, and the blobs in them\nfrom a remote repository using HTTP. It skips anything you already have.\n\nThere are a number of improvements possible, to be done if this catches\non, including, significantly, checking if the response was correct (or\neven not an error).\n\nIt makes fsck-cache and rev-tree give harmless warnings, because it\nincludes some code that should probably be shared with them in revision.h\n\nSigned-Off-By: Daniel Barkalow <barkalow@iabervon.org>\n\nIndex: Makefile\n===================================================================\n--- ed4f6e454b40650b904ab72048b2f93a068dccc3/Makefile  (mode:100644 sha1:b39b4ea37586693dd707d1d0750a9b580350ec50)\n+++ a65375b46154c90e7499b7e76998d430cd9cd29d/Makefile  (mode:100644 sha1:d41860aed161a14ca61e7b6c7f591f65928bd61f)\n@@ -14,7 +14,7 @@\n \n PROG=   update-cache show-diff init-db write-tree read-tree commit-tree \\\n \tcat-file fsck-cache checkout-cache diff-tree rev-tree show-files \\\n-\tcheck-files ls-tree merge-tree\n+\tcheck-files ls-tree merge-tree http-get\n \n all: $(PROG)\n \n@@ -23,6 +23,9 @@\n \n LIBS= -lssl -lz\n \n+http-get:%:%.o read-cache.o\n+\t$(CC) $(CFLAGS) -o $@ $^ $(LIBS)\n+\n init-db: init-db.o\n \n update-cache: update-cache.o read-cache.o\nIndex: http-get.c\n===================================================================\n--- /dev/null  (tree:ed4f6e454b40650b904ab72048b2f93a068dccc3)\n+++ a65375b46154c90e7499b7e76998d430cd9cd29d/http-get.c  (mode:100644 sha1:6a36cfa079519a7a3ad5b1618be8711c5127b531)\n@@ -0,0 +1,175 @@\n+#include <sys/socket.h>\n+#include <netdb.h>\n+#include <netinet/in.h>\n+#include <fcntl.h>\n+#include <unistd.h>\n+#include <string.h>\n+#include <stdlib.h>\n+#include \"cache.h\"\n+#include \"revision.h\"\n+#include <errno.h>\n+\n+static struct sockaddr_in sockad;\n+static char *url;\n+static char *base;\n+\n+static int target_url(char *target)\n+{\n+\tchar *name;\n+\tstruct hostent *entry;\n+\tif (memcmp(target, \"http://\", 7))\n+\t\treturn -1;\n+\turl = target;\n+\tbase = strchr(target + 7, '/');\n+\tname = malloc(base - (target + 7) + 1);\n+\tmemcpy(name, target + 7, base - (target + 7));\n+\tname[base - (target + 7)] = '\\0';\n+\tprintf(\"Connect to %s\\n\", name);\n+\tentry = gethostbyname(name);\n+\tmemcpy(&sockad.sin_addr.s_addr,\n+\t       &((struct in_addr *)entry->h_addr)->s_addr, 4);\n+\tsockad.sin_port = htons(80);\n+\tsockad.sin_family = AF_INET;\n+}\n+\n+static int get_connection()\n+{\n+\tint fd = socket(AF_INET, SOCK_STREAM, 0);\n+\tif (connect(fd, (struct sockaddr*) &sockad,\n+\t\t    sizeof(struct sockaddr_in))) {\n+\t\tperror(url);\n+\t}\n+\treturn fd;\n+}\n+\n+static void release_connection(int fd) {\n+\tclose(fd);\n+}\n+\n+static int fetch(unsigned char *sha1)\n+{\n+\tint header_end_posn = 0;\n+\tint local;\n+\tchar *hex = sha1_to_hex(sha1);\n+\tchar *filename = sha1_file_name(sha1);\n+\tchar buffer[4096];\n+\tint fd;\n+\tstruct stat st;\n+\n+\tif (!stat(filename, &st)) {\n+\t\treturn 0;\n+\t}\n+\n+\tfd = get_connection();\n+\tif (fd < 0) {\n+\t\treturn 1;\n+\t}\n+\n+\twrite(fd, \"GET \", 4);\n+\twrite(fd, base, strlen(base));\n+\twrite(fd, \"objects/\", 8);\n+\twrite(fd, hex, 2);\n+\twrite(fd, \"/\", 1);\n+\twrite(fd, hex + 2, 38);\n+\twrite(fd, \" HTTP/1.0\\r\\n\", 11);\n+\twrite(fd, \"\\r\\n\", 2);\n+\n+\tlocal = open(filename, O_WRONLY | O_CREAT | O_EXCL, 0666);\n+\n+\tdo {\n+\t\tint sz = read(fd, buffer, 4096);\n+\t\tif (!sz) {\n+\t\t\tbreak;\n+\t\t}\n+\t\tif (sz < 0) {\n+\t\t\tperror(\"Reading from connection\");\n+\t\t\tunlink(filename);\n+\t\t\tclose(local);\n+\t\t\treturn 1;\n+\t\t}\n+\t\tif (header_end_posn < 4) {\n+\t\t\tint i = 0;\n+\t\t\tchar *flag = \"\\r\\n\\r\\n\";\n+\t\t\twhile (i < sz && header_end_posn < 4) {\n+\t\t\t\tif (buffer[i] == flag[header_end_posn]) {\n+\t\t\t\t\theader_end_posn++;\n+\t\t\t\t} else {\n+\t\t\t\t\theader_end_posn = 0;\n+\t\t\t\t}\n+\t\t\t\ti++;\n+\t\t\t}\n+\t\t\tif (i < sz) {\n+\t\t\t\twrite(local, buffer + i, sz - i);\n+\t\t\t}\n+\t\t\tcontinue;\n+\t\t}\n+\t\twrite(local, buffer, sz);\n+\t} while (1);\n+\n+\tclose(local);\n+\t\n+\trelease_connection(fd);\n+\treturn 0;\n+}\n+\n+static int process_tree(unsigned char *sha1)\n+{\n+\tvoid *buffer;\n+        unsigned long size;\n+        char type[20];\n+\n+        buffer = read_sha1_file(sha1, type, &size);\n+\tif (!buffer)\n+\t\treturn -1;\n+\tif (strcmp(type, \"tree\"))\n+\t\treturn -1;\n+\twhile (size) {\n+\t\tint len = strlen(buffer) + 1;\n+\t\tunsigned char *sha1 = buffer + len;\n+\t\tunsigned int mode;\n+\t\tint retval;\n+\n+\t\tif (size < len + 20 || sscanf(buffer, \"%o\", &mode) != 1)\n+\t\t\treturn -1;\n+\n+\t\tbuffer = sha1 + 20;\n+\t\tsize -= len + 20;\n+\n+\t\tretval = fetch(sha1);\n+\t\tif (retval)\n+\t\t\treturn -1;\n+\n+\t\tif (S_ISDIR(mode)) {\n+\t\t\tretval = process_tree(sha1);\n+\t\t\tif (retval)\n+\t\t\t\treturn -1;\n+\t\t}\n+\t}\n+\treturn 0;\n+}\n+\n+static int process_commit(unsigned char *sha1)\n+{\n+\tstruct revision *rev = lookup_rev(sha1);\n+\tif (parse_commit_object(rev))\n+\t\treturn -1;\n+\t\n+\tfetch(rev->tree);\n+\tprocess_tree(rev->tree);\n+\treturn 0;\n+}\n+\n+int main(int argc, char **argv)\n+{\n+\tchar *commit_id = argv[1];\n+\tchar *url = argv[2];\n+\n+\tunsigned char sha1[20];\n+\n+\tget_sha1_hex(commit_id, sha1);\n+\n+\ttarget_url(url);\n+\n+\tfetch(sha1);\n+\treturn process_commit(sha1);\n+}\nIndex: revision.h\n===================================================================\n--- ed4f6e454b40650b904ab72048b2f93a068dccc3/revision.h  (mode:100664 sha1:28d0de3261a61f68e4e0948a25a416a515cd2e83)\n+++ a65375b46154c90e7499b7e76998d430cd9cd29d/revision.h  (mode:100664 sha1:523bde6e14e18bb0ecbded8f83ad4df93fc467ab)\n@@ -24,6 +24,7 @@\n \tunsigned int flags;\n \tunsigned char sha1[20];\n \tunsigned long date;\n+\tunsigned char tree[20];\n \tstruct parent *parent;\n };\n \n@@ -111,4 +112,29 @@\n \t}\n }\n \n+static int parse_commit_object(struct revision *rev)\n+{\n+\tif (!(rev->flags & SEEN)) {\n+\t\tvoid *buffer, *bufptr;\n+\t\tunsigned long size;\n+\t\tchar type[20];\n+\t\tunsigned char parent[20];\n+\n+\t\trev->flags |= SEEN;\n+\t\tbuffer = bufptr = read_sha1_file(rev->sha1, type, &size);\n+\t\tif (!buffer || strcmp(type, \"commit\"))\n+\t\t\treturn -1;\n+\t\tget_sha1_hex(bufptr + 5, rev->tree);\n+\t\tbufptr += 46; /* \"tree \" + \"hex sha1\" + \"\\n\" */\n+\t\twhile (!memcmp(bufptr, \"parent \", 7) && \n+\t\t       !get_sha1_hex(bufptr+7, parent)) {\n+\t\t\tadd_relationship(rev, parent);\n+\t\t\tbufptr += 48;   /* \"parent \" + \"hex sha1\" + \"\\n\" */\n+\t\t}\n+\t\t//rev->date = parse_commit_date(bufptr);\n+\t\tfree(buffer);\n+\t}\n+\treturn 0;\n+}\n+\n #endif /* REVISION_H */\n\n"},{"id":"358","messageId":"20050416221745.GA10280@ucw.cz","threadId":"54","inReplyTo":"Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Martin Mares","fromEmail":"mj@ucw.cz","sentAt":"2005-04-16T22:17:45Z","receivedAt":"2005-04-16T22:17:45Z","isPatch":true,"sender":{"key":"mj@ucw.cz","avatar":null},"body":"Hello!\n\n> This adds a program to download a commit, the trees, and the blobs in them\n> from a remote repository using HTTP. It skips anything you already have.\n\nIs it really necessary to write your own HTTP downloader? If so, is it\nnecessary to forget basic stuff like the \"Host:\" header? ;-)\n\nIf you feel that it should be optimized for speed, then at least use\npersistent connections.\n\n> +\tif (memcmp(target, \"http://\", 7))\n> +\t\treturn -1;\n\nCan crash if the string is too short.\n\n> +\tentry = gethostbyname(name);\n> +\tmemcpy(&sockad.sin_addr.s_addr,\n> +\t       &((struct in_addr *)entry->h_addr)->s_addr, 4);\n\nCan crash if the host doesn't exist or if you feed it with an URL containing\nport number.\n\n> +static int get_connection()\n\n(void)\n\n> +\tlocal = open(filename, O_WRONLY | O_CREAT | O_EXCL, 0666);\n\nWhat if it fails?\n\n\t\t\t\tHave a nice fortnight\n-- \nMartin `MJ' Mares   <mj@ucw.cz>   http://atrey.karlin.mff.cuni.cz/~mj/\nFaculty of Math and Physics, Charles University, Prague, Czech Rep., Earth\nA student who changes the course of history is probably taking an exam.\n"},{"id":"359","messageId":"12c511ca050416152452a4c620@mail.gmail.com","threadId":"54","inReplyTo":"Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Tony Luck","fromEmail":"tony.luck@gmail.com","sentAt":"2005-04-16T22:24:50Z","receivedAt":"2005-04-16T22:24:50Z","isPatch":true,"sender":{"key":"tony.luck@gmail.com","avatar":null},"body":"On 4/16/05, Daniel Barkalow <barkalow@iabervon.org> wrote:\n> +        buffer = read_sha1_file(sha1, type, &size);\n\nYou never free this buffer.\n\nIt would also be nice if you saved \"tree\" objects in some temporary file\nand did not install them until after you had fetched all the blobs and\ntrees that this tree references.  Then if your connection is interrupted\nyou can just restart it.\n\nOtherwise this looks really nice.  I was going to script something\nsimilar using \"wget\" ... but that would have made zillions of seperate\nconnections.  Not so kind to the server.\n\n-Tony\n"},{"id":"360","messageId":"20050416223206.GU9461@lug-owl.de","threadId":"54","inReplyTo":"Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Jan-Benedict Glaw","fromEmail":"jbglaw@lug-owl.de","sentAt":"2005-04-16T22:32:06Z","receivedAt":"2005-04-16T22:32:06Z","isPatch":true,"sender":{"key":"jbglaw@lug-owl.de","avatar":null},"body":"On Sat, 2005-04-16 18:03:51 -0400, Daniel Barkalow <barkalow@iabervon.org>\nwrote in message <Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org>:\n> --- /dev/null  (tree:ed4f6e454b40650b904ab72048b2f93a068dccc3)\n> +++ a65375b46154c90e7499b7e76998d430cd9cd29d/http-get.c  (mode:100644 sha1:6a36cfa079519a7a3ad5b1618be8711c5127b531)\n\n> +\tlocal = open(filename, O_WRONLY | O_CREAT | O_EXCL, 0666);\n\n0666 is a bit too lazy. I'd suggest 0664 or 0644.\n\nMfG, JBG\n\n-- \nJan-Benedict Glaw       jbglaw@lug-owl.de    . +49-172-7608481             _ O _\n\"Eine Freie Meinung in  einem Freien Kopf    | Gegen Zensur | Gegen Krieg  _ _ O\n fuer einen Freien Staat voll Freier Bürger\" | im Internet! |   im Irak!   O O O\nret = do_actions((curr | FREE_SPEECH) & ~(NEW_COPYRIGHT_LAW | DRM | TCPA));\n"},{"id":"361","messageId":"Pine.LNX.4.21.0504161827480.30848-100000@iabervon.org","threadId":"54","inReplyTo":"12c511ca050416152452a4c620@mail.gmail.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-04-16T22:33:18Z","receivedAt":"2005-04-16T22:33:18Z","isPatch":true,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Sat, 16 Apr 2005, Tony Luck wrote:\n\n> On 4/16/05, Daniel Barkalow <barkalow@iabervon.org> wrote:\n> > +        buffer = read_sha1_file(sha1, type, &size);\n> \n> You never free this buffer.\n\nIdeally, this should all be rearranged to share the code with\nread-tree, and it should be fixed in common.\n\n> It would also be nice if you saved \"tree\" objects in some temporary file\n> and did not install them until after you had fetched all the blobs and\n> trees that this tree references.  Then if your connection is interrupted\n> you can just restart it.\n\nIt looks over everything relevant, even if it doesn't need to download\nanything, so it should work to continue if it stops in between.\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"363","messageId":"Pine.LNX.4.21.0504161833340.30848-100000@iabervon.org","threadId":"54","inReplyTo":"20050416223206.GU9461@lug-owl.de","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-04-16T22:37:56Z","receivedAt":"2005-04-16T22:37:56Z","isPatch":true,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Sun, 17 Apr 2005, Jan-Benedict Glaw wrote:\n\n> On Sat, 2005-04-16 18:03:51 -0400, Daniel Barkalow <barkalow@iabervon.org>\n> wrote in message <Pine.LNX.4.21.0504161750020.30848-100000@iabervon.org>:\n> > --- /dev/null  (tree:ed4f6e454b40650b904ab72048b2f93a068dccc3)\n> > +++ a65375b46154c90e7499b7e76998d430cd9cd29d/http-get.c  (mode:100644 sha1:6a36cfa079519a7a3ad5b1618be8711c5127b531)\n> \n> > +\tlocal = open(filename, O_WRONLY | O_CREAT | O_EXCL, 0666);\n> \n> 0666 is a bit too lazy. I'd suggest 0664 or 0644.\n\nActually, 0444 would make most sense, since these shouldn't get modified\nat all. But umask is applied to them anyway, so 0664 or 0644 (or 0660 or \n0600) is up to the local system policy. This just matches\nwrite_sha1_buffer().\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"365","messageId":"011201c542d5$940bb670$03c8a8c0@kroptech.com","threadId":"54","inReplyTo":"12c511ca050416152452a4c620@mail.gmail.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Adam Kropelin","fromEmail":"akropel1@rochester.rr.com","sentAt":"2005-04-16T22:42:32Z","receivedAt":"2005-04-16T22:42:32Z","isPatch":true,"sender":{"key":"akropel1@rochester.rr.com","avatar":null},"body":"Tony Luck wrote:\n> Otherwise this looks really nice.  I was going to script something\n> similar using \"wget\" ... but that would have made zillions of seperate\n> connections.  Not so kind to the server.\n\nHow about building a file list and doing a batch download via 'wget -i \n/tmp/foo'? A quick test (on my ancient wget-1.7) indicates that it reuses \nconnectionss when successive URLs point to the same server.\n\nWriting yet another http client does seem a bit pointless, what with wget \nand curl available. The real win lies in creating the smarts to get the \nminimum number of files.\n\n--Adam\n\n"},{"id":"366","messageId":"Pine.LNX.4.21.0504161838030.30848-100000@iabervon.org","threadId":"54","inReplyTo":"20050416221745.GA10280@ucw.cz","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-04-16T22:43:42Z","receivedAt":"2005-04-16T22:43:42Z","isPatch":true,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Sun, 17 Apr 2005, Martin Mares wrote:\n\n> Hello!\n> \n> > This adds a program to download a commit, the trees, and the blobs in them\n> > from a remote repository using HTTP. It skips anything you already have.\n> \n> Is it really necessary to write your own HTTP downloader? If so, is it\n> necessary to forget basic stuff like the \"Host:\" header? ;-)\n\nI wanted to get something hacked quickly; can you suggest a good one to\nuse?\n\n> If you feel that it should be optimized for speed, then at least use\n> persistent connections.\n\nThat's the next step.\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"367","messageId":"Pine.LNX.4.21.0504161844040.30848-100000@iabervon.org","threadId":"54","inReplyTo":"011201c542d5$940bb670$03c8a8c0@kroptech.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-04-16T22:45:54Z","receivedAt":"2005-04-16T22:45:54Z","isPatch":true,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Sat, 16 Apr 2005, Adam Kropelin wrote:\n\n> Tony Luck wrote:\n> > Otherwise this looks really nice.  I was going to script something\n> > similar using \"wget\" ... but that would have made zillions of seperate\n> > connections.  Not so kind to the server.\n> \n> How about building a file list and doing a batch download via 'wget -i \n> /tmp/foo'? A quick test (on my ancient wget-1.7) indicates that it reuses \n> connectionss when successive URLs point to the same server.\n\nYou need to look at some of the files before you know what other files to\nget. You could do it in waves, but that would be excessively complicated\nto code and not the most efficient anyway.\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"395","messageId":"014a01c542d6$ff3e6180$03c8a8c0@kroptech.com","threadId":"54","inReplyTo":"Pine.LNX.4.21.0504161844040.30848-100000@iabervon.org","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Adam Kropelin","fromEmail":"akropel1@rochester.rr.com","sentAt":"2005-04-16T22:52:42Z","receivedAt":"2005-04-16T22:52:42Z","isPatch":true,"sender":{"key":"akropel1@rochester.rr.com","avatar":null},"body":"Daniel Barkalow wrote:\n> On Sat, 16 Apr 2005, Adam Kropelin wrote:\n>> How about building a file list and doing a batch download via 'wget\n>> -i /tmp/foo'? A quick test (on my ancient wget-1.7) indicates that\n>> it reuses connectionss when successive URLs point to the same server.\n>\n> You need to look at some of the files before you know what other\n> files to get. You could do it in waves, but that would be excessively\n> complicated to code and not the most efficient anyway.\n\nAh, yes. Makes sense. How about libcurl or another http client library, \nthen? Minimizing dependencies on external libraries is good, but writing a \nreally robust http client is a tricky business. (Not that you aren't up to \nit; I just wonder if it's the best way to spend your time.)\n\n--Adam\n\n"},{"id":"422","messageId":"200504170316.j3H3GaZ03333@unix-os.sc.intel.com","threadId":"54","inReplyTo":"011201c542d5$940bb670$03c8a8c0@kroptech.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"","fromEmail":"tony.luck@intel.com","sentAt":"2005-04-17T03:16:36Z","receivedAt":"2005-04-17T03:16:36Z","isPatch":true,"sender":{"key":"tony.luck@intel.com","avatar":"https://avatars.githubusercontent.com/u/5446021?v=4"},"body":">How about building a file list and doing a batch download via 'wget -i \n>/tmp/foo'? A quick test (on my ancient wget-1.7) indicates that it reuses \n>connectionss when successive URLs point to the same server.\n\nHere's a script that does just that.  So there is a burst of individual\nwget commands to get HEAD, the top commit object, and all the tree\nobjects.  The just one to get all the missing blobs.\n\nSubsequent runs will do far less work as many of the tree objects will\nnot have changed, so we don't descend into any tree that we already have.\n\n-Tony\n\nNot a patch ... it is a whole file.  I called it \"git-wget\", but it might\nalso want to be called \"git-pulltop\".\n\nSigned-off-by: Tony Luck <tony.luck@intel.com>\n\n------ script starts here -----\n#!/bin/sh\n\n# Copyright (C) 2005 Tony Luck\n\nREMOTE=http://www.kernel.org/pub/linux/kernel/people/torvalds/linux-2.6.git/\n\nrm -rf .gittmp\n# set up a temp git repository so that we can use cat-file and ls-tree on the\n# objects we pull without installing them into our tree. This allows us to\n# restart if the download is interrupted\nmkdir .gittmp\ncd .gittmp\ninit-db\n\nwget -q $REMOTE/HEAD\n\nif cmp -s ../.git/HEAD HEAD\nthen\n\techo Already have HEAD = `cat ../.git/HEAD`\n\tcd ..\n\trm -rf .gittmp\n\texit 0\nfi\n\nsha1=`cat HEAD`\nsha1file=${sha1:0:2}/${sha1:2}\n\nif [ -f ../.git/objects/$sha1file ]\nthen\n\techo Already have most recent commit. Update HEAD to $sha1\n\tcd ..\n\trm -rf .gittmp\n\texit 0\nfi\n\nwget -q $REMOTE/objects/$sha1file -O .git/objects/$sha1file\n\ntreesha1=`cat-file commit $sha1 | (read tag tree ; echo $tree)`\n\nget_tree()\n{\n\ttreesha1file=${1:0:2}/${1:2}\n\tif [ -f ../.git/objects/$treesha1file ]\n\tthen\n\t\treturn\n\tfi\n\twget -q $REMOTE/objects/$treesha1file -O .git/objects/$treesha1file\n\tls-tree $1 | while read mode tag sha1 name\n\tdo\n\t\tsubsha1file=${sha1:0:2}/${sha1:2}\n\t\tif [  -f ../.git/objects/$subsha1file ]\n\t\tthen\n\t\t\tcontinue\n\t\tfi\n\t\tif [ $mode = 40000 ]\n\t\tthen\n\t\t\tget_tree $sha1 `expr $2 + 1`\n\t\telse\n\t\t\techo objects/$subsha1file >> needbloblist\n\t\tfi\n\tdone\n}\n\n# get all the tree objects to our .gittmp area, and create list of needed blobs\nget_tree $treesha1\n\n# now get the blobs\ncd ../.git\nif [ -s ../.gittmp/needbloblist ]\nthen\n\twget -q -r -nH  --cut-dirs=6 --base=$REMOTE -i ../.gittmp/needbloblist\nfi\n\n# Now we have the blobs, move the trees and commit from .gitttmp\ncd ../.gittmp/.git/objects\nfind ?? -type f -print | while read f\ndo\n\tmv $f ../../../.git/objects/$f\ndone\n\n# update HEAD\ncd ../..\nmv HEAD ../.git\n\ncd ..\nrm -rf .gittmp\n------ script ends here -----\n"},{"id":"678","messageId":"200504181841.j3IIfgP31258@unix-os.sc.intel.com","threadId":"54","inReplyTo":"200504170316.j3H3GaZ03333@unix-os.sc.intel.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"","fromEmail":"tony.luck@intel.com","sentAt":"2005-04-18T18:41:42Z","receivedAt":"2005-04-18T18:41:42Z","isPatch":true,"sender":{"key":"tony.luck@intel.com","avatar":"https://avatars.githubusercontent.com/u/5446021?v=4"},"body":">Not a patch ... it is a whole file.  I called it \"git-wget\", but it might\n>also want to be called \"git-pulltop\".\n\nIt's been pointed out to me that I based this script on a pre-historic version\nof ls-tree from sometime last week.  Modern versions print the mode with %06o\nso there is a leading 0 on the mode for a directory.  Just change\n\n\t\tif [ $mode = 40000 ]\n\nto\n\n\t\tif [ $mode = 040000 ]\n\nto fix it.\n\nThe script might also be useful for anyone behind a firewall that blocks\nrsync transfers.\n\n-Tony\n"},{"id":"679","messageId":"20050418184750.GD5554@pasky.ji.cz","threadId":"54","inReplyTo":"200504181841.j3IIfgP31258@unix-os.sc.intel.com","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"Petr Baudis","fromEmail":"pasky@ucw.cz","sentAt":"2005-04-18T18:47:50Z","receivedAt":"2005-04-18T18:47:50Z","isPatch":true,"sender":{"key":"pasky@ucw.cz","avatar":"https://avatars.githubusercontent.com/u/18439?v=4"},"body":"Dear diary, on Mon, Apr 18, 2005 at 08:41:42PM CEST, I got a letter\nwhere tony.luck@intel.com told me that...\n> >Not a patch ... it is a whole file.  I called it \"git-wget\", but it might\n> >also want to be called \"git-pulltop\".\n> \n> It's been pointed out to me that I based this script on a pre-historic version\n> of ls-tree from sometime last week.  Modern versions print the mode with %06o\n> so there is a leading 0 on the mode for a directory.  Just change\n> \n> \t\tif [ $mode = 40000 ]\n> \n> to\n> \n> \t\tif [ $mode = 040000 ]\n> \n> to fix it.\n\n...and this is precisely why ls-tree actually outputs those \"blob\" and\n\"tree\" tags. ;-)\n\n-- \n\t\t\t\tPetr \"Pasky\" Baudis\nStuff: http://pasky.or.cz/\nC++: an octopus made by nailing extra legs onto a dog. -- Steve Taylor\n"},{"id":"694","messageId":"200504182048.j3IKmoi32162@unix-os.sc.intel.com","threadId":"54","inReplyTo":"20050418184750.GD5554@pasky.ji.cz","subject":"Re: [PATCH] Get commits from remote repositories by HTTP","fromName":"","fromEmail":"tony.luck@intel.com","sentAt":"2005-04-18T20:48:50Z","receivedAt":"2005-04-18T20:48:50Z","isPatch":true,"sender":{"key":"tony.luck@intel.com","avatar":"https://avatars.githubusercontent.com/u/5446021?v=4"},"body":"> ...and this is precisely why ls-tree actually outputs those \"blob\" and\n> \"tree\" tags. ;-)\n\nDoh!\n\nHere's a fresh copy with \"if [ $tag = tree ]\".  I just used it to pull\nfrom Linus into an \"empty\" directory (just ran init-db to make the .git\n.git/objects and .git/objects/xx directories).\n\n-Tony\n\n\n#!/bin/bash\n\n# Copyright (C) 2005 Tony Luck\n\nREMOTE=http://www.kernel.org/pub/linux/kernel/people/torvalds/linux-2.6.git/\n\nrm -rf .gittmp\n# set up a temp git repository so that we can use cat-file and ls-tree on the\n# objects we pull without installing them into our tree. This allows us to\n# restart if the download is interrupted\nmkdir .gittmp\ncd .gittmp\ninit-db\n\nwget -q $REMOTE/HEAD\n\nif cmp -s ../.git/HEAD HEAD\nthen\n\techo Already have HEAD = `cat ../.git/HEAD`\n\tcd ..\n\trm -rf .gittmp\n\texit 0\nfi\n\nsha1=`cat HEAD`\nsha1file=${sha1:0:2}/${sha1:2}\n\nif [ -f ../.git/objects/$sha1file ]\nthen\n\techo Already have most recent commit. Update HEAD to $sha1\n\tcd ..\n\trm -rf .gittmp\n\texit 0\nfi\n\nwget -q $REMOTE/objects/$sha1file -O .git/objects/$sha1file\n\ntreesha1=`cat-file commit $sha1 | (read tag tree ; echo $tree)`\n\nget_tree()\n{\n\ttreesha1file=${1:0:2}/${1:2}\n\tif [ -f ../.git/objects/$treesha1file ]\n\tthen\n\t\treturn\n\tfi\n\twget -q $REMOTE/objects/$treesha1file -O .git/objects/$treesha1file\n\tls-tree $1 | while read mode tag sha1 name\n\tdo\n\t\tsubsha1file=${sha1:0:2}/${sha1:2}\n\t\tif [  -f ../.git/objects/$subsha1file ]\n\t\tthen\n\t\t\tcontinue\n\t\tfi\n\t\tif [ $tag = tree ]\n\t\tthen\n\t\t\tget_tree $sha1 `expr $2 + 1`\n\t\telse\n\t\t\techo objects/$subsha1file >> needbloblist\n\t\tfi\n\tdone\n}\n\n# get all the tree objects to our .gittmp area, and create list of needed blobs\nget_tree $treesha1\n\n# now get the blobs\ncd ../.git\nif [ -s ../.gittmp/needbloblist ]\nthen\n\twget -q -r -nH  --cut-dirs=6 --base=$REMOTE -i ../.gittmp/needbloblist\nfi\n\n# Now we have the blobs, move the trees and commit from .gitttmp\ncd ../.gittmp/.git/objects\nfind ?? -type f -print | while read f\ndo\n\tmv $f ../../../.git/objects/$f\ndone\n\n# update HEAD\ncd ../..\nmv HEAD ../.git\n\ncd ..\nrm -rf .gittmp\n"}]}