{"thread":{"id":"779","subject":"[SCRIPT] cg-rpush & locking","startedAt":"2005-05-31T19:00:05Z","lastAt":"2005-06-02T19:15:21Z","messageCount":13,"participants":["Tony Lindgren","Nicolas Pitre","Thomas Glanzmann","Linus Torvalds","Daniel Barkalow","Matthias Urlichs","Dan Holmsand"],"isPatch":false,"patchVersion":null,"patchTotal":null},"messages":[{"id":"4314","messageId":"20050531190005.GE18723@atomide.com","threadId":"779","inReplyTo":null,"subject":"[SCRIPT] cg-rpush & locking","fromName":"Tony Lindgren","fromEmail":"tony@atomide.com","sentAt":"2005-05-31T19:00:05Z","receivedAt":"2005-05-31T19:00:05Z","isPatch":false,"sender":{"key":"tony@atomide.com","avatar":"https://gravatar.com/avatar/d536761d645f787e43f0491b2c7c003f9e3c1041a5458caafce7e5b5618a6530?d=mp&s=160"},"body":"Hello all,\n\nAttached is a little script we're using for pushing changes to the\nlinux-omap tree. I modified it from an earlier script done by\nMatthias Urlichs.\n\nIt uses rsync over ssh and should work with rsync write access too.\n\nIn order to remove lock files using rsync, I've added\na .git/locks subdirectory that contains the lock file.\n\nThe reason for this is that it allows replacing the lock directory\nwith an empty directory using rsync. This removes the lock file\nafter the remote update is done.\n\nCurrently the local lock has some issues with chained pushes...\nIf somebody is pushing from a remote repo to the local repo while\nthe local repo is being pushed to some other remote repo, the lock\nmay get removed.\n\nOf course the lock does not protect from local changes either.\n\nAnybody have any better ideas for locking that also works with\nrsync?\n\nTony\n\n\n#!/bin/sh\n#\n# Pushes changes from the local git repo to remote repo.\n#\n# Copyright (C) 2005 Tony Lindgren <tony@atomide.com>\n#\n# Some parts based on an earlier push script, \n# Copyright (C) 2005 Matthias Urlichs.\n#\n# Takes the remote repo's name or url as parameter.\n#\n# Most likely the remote repo is not rsync writable, but\n# you can use rsync over ssh for the push.\n#\n# When using rsync over ssh, you must use the real repo\n# path on the server and an ssh key to avoid typing in\n# the password multiple times. For example:\n#\n# $ export RSYNC_FLAGS=\"-z --progress\"\n# $ export RSYNC_RSH=\"ssh -i /home/user/.ssh/my-git-key\"\n# $ cg-rpush some.machine:/home/git/repo\n#\n\n. ${COGITO_LIB}cg-Xlib\n\nname=$1\n\nBRANCHES=\".git/branches\"\nHEADS=\".git/refs/heads\"\nOBJECTS=\".git/objects\"\nLOCKS=\".git/locks\"\nREMOTE_LOCK=\"write_lock\"\nTMP=\"/tmp\"\n\nuri=\"\"\n\nfunction usage () {\n\techo \"Usage: [RSYNC_FLAGS=\\\"-e ssh\\\"] $0 some.machine:/home/git/repo\"\n\texit 1\n}\n\nfunction die () {\n\tif [ -f $LOCKS/$REMOTE_LOCK ]; then\n\t\trm -f $LOCKS/$REMOTE_LOCK\n\tfi\n\techo cg-rpush: $@ >&2\n\texit 1\n}\n\nfunction clean_locks () {\n\trm -f $LOCKS/$REMOTE_LOCK\n\trsync $RSYNC_FLAGS -r --delete $LOCKS/ $uri/$LOCKS\n}\n\nfunction validate_input () {\n\t[ \"$name\" ] || usage\n\tif [ ! -d \".git\" ]; then\n\t\tdie \"Could not find local .git directory\"\n\tfi\n\turi=$(cat $BRANCHES/$name 2>/dev/null)\n\t[ \"$uri\" ] || uri=$name\n\techo $uri\n}\n\n#\n# We must use a lock directory to allow removing the remote lock\n# files with rsync by copying over it with an empty directory.\n# Creating the remote lock file should be safe. However, please note\n# that we must also be careful not to remove local .git/locks/write_lock\n# in case somebody is pushing to our local repo from a remote machine.\n# Currently the local lock file creation can conflict with a lock\n# file creation from a remote machine to our local machine.\n#\nfunction lock_files () {\n\techo \"Attempting to to create a write lock on remote...\"\n\tif [ ! -d $LOCKS ]; then\n\t\tmkdir $LOCKS;\n\tfi\n\tif [ -f $LOCKS/$REMOTE_LOCK ]; then\n\t\techo \"Local write_lock already exists: $LOCKS/$REMOTE_LOCK\"\n\t\texit 1\n\tfi\n\tlock_stamp=\"$USER@$HOSTNAME $(date)\"\n\techo $lock_stamp > $LOCKS/$REMOTE_LOCK\n\trsync $RSYNC_FLAGS -r --ignore-existing $LOCKS/ $uri/$LOCKS\n\n\t# Check what the remote .git/locks/write_lock has\n\ttmpfile=$TMP/remote_lock_$RANDOM\n\trsync $RSYNC_FLAGS \"$uri/$LOCKS/$REMOTE_LOCK\" $tmpfile\n\tremote_stamp=$(cat $tmpfile)\n\trm -f $tmpfile\n\tif [ \"$remote_stamp\" != \"$lock_stamp\" ]; then\n\t\tdie \"Remote locked by $remote_stamp, please try again later\"\n\tfi\n}\n\nfunction check_remote_version () {\n\techo \"Getting remote version...\"\n\ttmpfile=$TMP/remote_head_$RANDOM\n\trsync $RSYNC_FLAGS -Lr \"$uri/$HEADS/master\" $tmpfile\n\tremote_head=$(cat $tmpfile)\n\trm -f $tmpfile\n\tif [ -z \"$remote_head\" ]; then\n\t\tclean_locks\n\t\tdie \"Remote repository does not have $uri/$HEADS/master\"\n\tfi\n\techo \"Remote head is at: $remote_head\"\n\tif [ \"$(git-cat-file -t \"$remote_head\" 2>/dev/null)\" != \"commit\" ]; then\n\t\tclean_locks\n\t\tdie \"Remote is ahead, please do a pull first\"\n\tfi\n}\n\nfunction push_git_objects () {\n\techo \"Pushing .git/objects...\"\n\trsync $RSYNC_FLAGS --ignore-existing --whole-file -v -r \\\n\t\t\"$OBJECTS/\" \"$uri/$OBJECTS/\"\n}\n\nfunction update_remote_head () {\n\tlocal_head=$(cat $HEADS/master)\n\techo \"Updating remote head to: $local_head\"\n\trsync $RSYNC_FLAGS -Lr $HEADS/master \"$uri/$HEADS/master\"\n}\n\nfunction print_note () {\n\techo \"Remote updated successfully\"\n\techo \"NOTE: Not updating checked out remote files in case they\"\n\techo \"have been edited locally on the remote machine.\"\n\techo \"To sync checked out files on remote, you can run cg-cancel\"\n\techo \"on remote machine. You can check for uncommitted changes\"\n\techo \"on remote with cg-diff first, which should only show\"\n\techo \"changes done in this push.\"\n}\n\n#\n# Main program\n#\nuri=$(validate_input)\nlock_files\ncheck_remote_version\npush_git_objects\nupdate_remote_head\nclean_locks\nprint_note\nexit\n"},{"id":"4328","messageId":"Pine.LNX.4.63.0505311914550.6500@localhost.localdomain","threadId":"779","inReplyTo":"20050531190005.GE18723@atomide.com","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Nicolas Pitre","fromEmail":"nico@cam.org","sentAt":"2005-05-31T23:16:09Z","receivedAt":"2005-05-31T23:16:09Z","isPatch":false,"sender":{"key":"nico@fluxnic.net","avatar":"https://avatars.githubusercontent.com/u/702790?v=4"},"body":"On Tue, 31 May 2005, Tony Lindgren wrote:\n\n> Anybody have any better ideas for locking that also works with\n> rsync?\n\nWhy do you need a lock at all?\n\nJust update your HEAD reference last when you push and get it first when \nyou pull.\n\n\nNicolas\n"},{"id":"4355","messageId":"20050601065123.GA23358@cip.informatik.uni-erlangen.de","threadId":"779","inReplyTo":"Pine.LNX.4.63.0505311914550.6500@localhost.localdomain","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Thomas Glanzmann","fromEmail":"sithglan@stud.uni-erlangen.de","sentAt":"2005-06-01T06:51:23Z","receivedAt":"2005-06-01T06:51:23Z","isPatch":false,"sender":{"key":"sithglan@stud.uni-erlangen.de","avatar":null},"body":"Hello,\n\n> Why do you need a lock at all?\n\n> Just update your HEAD reference last when you push and get it first when \n> you pull.\n\nconsider the following scenario: Two people push at the same time. One\nHEAD gets actually written, but both think that their changes got\nupstream. Of course the 'upstream' tree is consitent, but incomplete.\nThat is why we need a lock. And the lock should be obtained before the\nremote HEAD is retrieved, I think the following scenario is how to\nhandle it:\n\n\t1. acquire remote lock\n\t2. get remote HEAD\n\t3. if remote HEAD is ahead (not included in our history) abort\n\t   and free lock.\n\t4. push objects\n\t5. update remote HEAD with local\n\t6. free remote lock.\n\n\tThomas\n"},{"id":"4375","messageId":"20050601165502.GB20936@atomide.com","threadId":"779","inReplyTo":"20050601065123.GA23358@cip.informatik.uni-erlangen.de","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Tony Lindgren","fromEmail":"tony@atomide.com","sentAt":"2005-06-01T16:55:02Z","receivedAt":"2005-06-01T16:55:02Z","isPatch":false,"sender":{"key":"tony@atomide.com","avatar":"https://gravatar.com/avatar/d536761d645f787e43f0491b2c7c003f9e3c1041a5458caafce7e5b5618a6530?d=mp&s=160"},"body":"* Thomas Glanzmann <sithglan@stud.uni-erlangen.de> [050531 23:56]:\n> Hello,\n> \n> > Why do you need a lock at all?\n> \n> > Just update your HEAD reference last when you push and get it first when \n> > you pull.\n> \n> consider the following scenario: Two people push at the same time. One\n> HEAD gets actually written, but both think that their changes got\n> upstream. Of course the 'upstream' tree is consitent, but incomplete.\n> That is why we need a lock. And the lock should be obtained before the\n> remote HEAD is retrieved, I think the following scenario is how to\n> handle it:\n> \n> \t1. acquire remote lock\n> \t2. get remote HEAD\n> \t3. if remote HEAD is ahead (not included in our history) abort\n> \t   and free lock.\n> \t4. push objects\n> \t5. update remote HEAD with local\n> \t6. free remote lock.\n\nYes, that's basically what the script does. We have several people\ncommitting patches.\n\nTony\n"},{"id":"4396","messageId":"Pine.LNX.4.58.0506011951150.1876@ppc970.osdl.org","threadId":"779","inReplyTo":"20050601065123.GA23358@cip.informatik.uni-erlangen.de","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Linus Torvalds","fromEmail":"torvalds@osdl.org","sentAt":"2005-06-02T02:58:59Z","receivedAt":"2005-06-02T02:58:59Z","isPatch":false,"sender":{"key":"torvalds@linux-foundation.org","avatar":"https://avatars.githubusercontent.com/u/1024025?v=4"},"body":"\n\nOn Wed, 1 Jun 2005, Thomas Glanzmann wrote:\n> \n> \t1. acquire remote lock\n> \t2. get remote HEAD\n> \t3. if remote HEAD is ahead (not included in our history) abort\n> \t   and free lock.\n> \t4. push objects\n> \t5. update remote HEAD with local\n> \t6. free remote lock.\n\nYou really need a specialized client at the other end, because regardless \nof locking, you want to write the objects atomically (ie download them \ninto a temp-file, and then do the \"rename\" thing to make them show up \nall-or-nothing).\n\nAlso, I'd suggest a slight modification to avoid keeping the lock for a \nlong time, namely to have the lock protect just a quick \"compare and \nexchange\". So the algorithm would become:\n\n\t1. read remote HEAD\n\t2. if remote HEAD isn't in our history, abort with \"remote is \n\t   ahead\"\n\t3. calculate the objects needed to push locally\n\t4. push them (but accept the possibility that the remote may\n\t   already have them, so have the protocol able to say \"got that\n\t   one already\"). Make this use the atomic write on the other end.\n\t5. do an atomic compare-and-exchange of the remote head with the \n\t   new one (ie only switch the remote HEAD if it still matches \n\t   what we were expecting it to be)\n\nHmm?\n\n\t\tLinus\n"},{"id":"4399","messageId":"Pine.LNX.4.21.0506020223570.30848-100000@iabervon.org","threadId":"779","inReplyTo":"Pine.LNX.4.58.0506011951150.1876@ppc970.osdl.org","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-06-02T06:39:07Z","receivedAt":"2005-06-02T06:39:07Z","isPatch":false,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Wed, 1 Jun 2005, Linus Torvalds wrote:\n\n> \n> \n> On Wed, 1 Jun 2005, Thomas Glanzmann wrote:\n> > \n> > \t1. acquire remote lock\n> > \t2. get remote HEAD\n> > \t3. if remote HEAD is ahead (not included in our history) abort\n> > \t   and free lock.\n> > \t4. push objects\n> > \t5. update remote HEAD with local\n> > \t6. free remote lock.\n> \n> You really need a specialized client at the other end, because regardless \n> of locking, you want to write the objects atomically (ie download them \n> into a temp-file, and then do the \"rename\" thing to make them show up \n> all-or-nothing).\n> \n> Also, I'd suggest a slight modification to avoid keeping the lock for a \n> long time, namely to have the lock protect just a quick \"compare and \n> exchange\". So the algorithm would become:\n> \n> \t1. read remote HEAD\n> \t2. if remote HEAD isn't in our history, abort with \"remote is \n> \t   ahead\"\n> \t3. calculate the objects needed to push locally\n> \t4. push them (but accept the possibility that the remote may\n> \t   already have them, so have the protocol able to say \"got that\n> \t   one already\"). Make this use the atomic write on the other end.\n> \t5. do an atomic compare-and-exchange of the remote head with the \n> \t   new one (ie only switch the remote HEAD if it still matches \n> \t   what we were expecting it to be)\n> \n> Hmm?\n\nIf the lock is only to protect against someone else modifying HEAD after\nwe've checked that it is our starting point and before we modify it,\nthere's no reason not to hold the lock while pushing; it wouldn't block\nanything other than someone doing a quick push in the middle of our long\none, and thereby causing us to dump a lot of useless objects on the\nserver (which will become obsolete as we will need to do the merge and\npush a different version).\n\nThe key is that people can still download the old version until the new\nversion is there, regardless of the lock; they'll get data about to\ngo stale, but they would have anyway had they been a few seconds\nearlier. The main annoyance would be that you'd be blocked from pushing,\nand then have to poll for the other upload to finish before you'd be able\nto pull, do the merge, and then push your changes; you want to have the\nclient watch for the resolution of the other transfer one way or the\nother, since you're in the current state precisely because you lost on\ngetting the lock and now definitely need the next version.\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"4400","messageId":"20050602071453.GA16616@kiste.smurf.noris.de","threadId":"779","inReplyTo":"Pine.LNX.4.21.0506020223570.30848-100000@iabervon.org","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Matthias Urlichs","fromEmail":"smurf@smurf.noris.de","sentAt":"2005-06-02T07:14:53Z","receivedAt":"2005-06-02T07:14:53Z","isPatch":false,"sender":{"key":"matthias@urlichs.de","avatar":"https://gravatar.com/avatar/2708905af227313eba6f2b2ae0f7d0259b5ac5d71baef58fe5a13c699ce0bbf0?d=mp&s=160"},"body":"Hi,\n\nDaniel Barkalow:\n> If the lock is only to protect against someone else modifying HEAD after\n> we've checked that it is our starting point and before we modify it,\n> there's no reason not to hold the lock while pushing; it wouldn't block\n> anything other than someone doing a quick push in the middle of our long\n> one, and thereby causing us to dump a lot of useless objects on the\n> server (which will become obsolete as we will need to do the merge and\n> push a different version).\n> \nThe objects we push aren't going to be obsolete. The server needs them\nanyway, because our HEAD refers to them.\n\nWhat if the connection dies in the middle of a push? You then sit there\nwaiting for it, and the lock, to time out. OTOH, an atomic cmpxchg on\nthe server can't block and can't timeout.\n\n> you want to have the\n> client watch for the resolution of the other transfer one way or the\n> other, since you're in the current state precisely because you lost on\n> getting the lock and now definitely need the next version.\n> \nI disagree. Given that you need to wait for the upload to finish anyway\n(whether you know it or not ;-) it makes sense to spend the time\nactually uploading -- upload speed is frequently lower than download\nfor individuals.\n\n-- \nMatthias Urlichs   |   {M:U} IT Design @ m-u-it.de   |  smurf@smurf.noris.de\nDisclaimer: The quote was selected randomly. Really. | http://smurf.noris.de\n - -\nWho was that masked man?\n"},{"id":"4402","messageId":"20050602073205.GA31482@muru.com","threadId":"779","inReplyTo":"20050602071453.GA16616@kiste.smurf.noris.de","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Tony Lindgren","fromEmail":"tony@atomide.com","sentAt":"2005-06-02T07:32:05Z","receivedAt":"2005-06-02T07:32:05Z","isPatch":false,"sender":{"key":"tony@atomide.com","avatar":"https://gravatar.com/avatar/d536761d645f787e43f0491b2c7c003f9e3c1041a5458caafce7e5b5618a6530?d=mp&s=160"},"body":"On Thu, Jun 02, 2005 at 09:14:53AM +0200, Matthias Urlichs wrote:\n> Hi,\n> \n> Daniel Barkalow:\n> > If the lock is only to protect against someone else modifying HEAD after\n> > we've checked that it is our starting point and before we modify it,\n> > there's no reason not to hold the lock while pushing; it wouldn't block\n> > anything other than someone doing a quick push in the middle of our long\n> > one, and thereby causing us to dump a lot of useless objects on the\n> > server (which will become obsolete as we will need to do the merge and\n> > push a different version).\n> > \n> The objects we push aren't going to be obsolete. The server needs them\n> anyway, because our HEAD refers to them.\n\nI don't think locking for the duration of the push really is a problem.\nIt is unlikely that there would be so many people pushing that it would\ncause inconvenience... Of course it would be nice to optimize it if\npossible.\n\n> What if the connection dies in the middle of a push? You then sit there\n> waiting for it, and the lock, to time out. OTOH, an atomic cmpxchg on\n> the server can't block and can't timeout.\n> \n> > you want to have the\n> > client watch for the resolution of the other transfer one way or the\n> > other, since you're in the current state precisely because you lost on\n> > getting the lock and now definitely need the next version.\n> > \n> I disagree. Given that you need to wait for the upload to finish anyway\n> (whether you know it or not ;-) it makes sense to spend the time\n> actually uploading -- upload speed is frequently lower than download\n> for individuals.\n\nI would assume the biggest problem for most people is how they can push\nthrough a firewall. From that point of view it would make sense to do\nthe push as a cgi script rather than something over ssh. And with a\ncgi script you can of course optimize the locking and use tmp files\nbefore renaming which are a bit hard to do with rsync.\n\nTony\n"},{"id":"4406","messageId":"20050602100417.GG16616@kiste.smurf.noris.de","threadId":"779","inReplyTo":"20050602073205.GA31482@muru.com","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Matthias Urlichs","fromEmail":"smurf@smurf.noris.de","sentAt":"2005-06-02T10:04:17Z","receivedAt":"2005-06-02T10:04:17Z","isPatch":false,"sender":{"key":"matthias@urlichs.de","avatar":"https://gravatar.com/avatar/2708905af227313eba6f2b2ae0f7d0259b5ac5d71baef58fe5a13c699ce0bbf0?d=mp&s=160"},"body":"Hi,\n\nTony Lindgren:\n> I don't think locking for the duration of the push really is a problem.\n> It is unlikely that there would be so many people pushing that it would\n> cause inconvenience... Of course it would be nice to optimize it if\n> possible.\n> \nSince an 'atomic' cmpxchg operation is actually easier to program than a\nlockfile with timeout handling etc., I would hope so. ;-)\n\n> I would assume the biggest problem for most people is how they can push\n> through a firewall. From that point of view it would make sense to do\n> the push as a cgi script rather than something over ssh.\n\nWhen in doubt, do both ... I'd certainly prefer ssh if at all possible.\n\n> And with a cgi script you can of course optimize the locking and use\n> tmp files before renaming which are a bit hard to do with rsync.\n> \nrsync is going away anyway for git usage (long-term), I'd assume. It\ncertainly becomes more and more ineffective the more our history is\ngrowing.\n\n-- \nMatthias Urlichs   |   {M:U} IT Design @ m-u-it.de   |  smurf@smurf.noris.de\nDisclaimer: The quote was selected randomly. Really. | http://smurf.noris.de\n - -\nNo one can guarantee the actions of another.\n\t\t-- Spock, \"Day of the Dove\", stardate unknown\n"},{"id":"4418","messageId":"Pine.LNX.4.58.0506020741480.1876@ppc970.osdl.org","threadId":"779","inReplyTo":"20050602073205.GA31482@muru.com","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Linus Torvalds","fromEmail":"torvalds@osdl.org","sentAt":"2005-06-02T14:50:27Z","receivedAt":"2005-06-02T14:50:27Z","isPatch":false,"sender":{"key":"torvalds@linux-foundation.org","avatar":"https://avatars.githubusercontent.com/u/1024025?v=4"},"body":"\n\nOn Thu, 2 Jun 2005, Tony Lindgren wrote:\n> \n> I don't think locking for the duration of the push really is a problem.\n> It is unlikely that there would be so many people pushing that it would\n> cause inconvenience... Of course it would be nice to optimize it if\n> possible.\n\nI don't think the locking is a _huge_ issue - the only real problem I had\nwith locking with BK was that readers could block a writer (ie I couldn't\npush to my public tree if there were people downloading from it), and git\ndoesn't have that problem. I only push to trees that are my private ones\nanyway.\n\nThe real reason I'd prefer to not do locking is that _if_ the remote tree\nis actually more than just a CVS-like \"public repository\", ie if somebody\nactually does _development_ in the remote tree (hey, it may be crazy, but\ngit makes this usage pattern possible), then we should eventually plan on\nhaving all of the regular \"git-commit-script\" and \"git-pull-script\" etc\n_also_ do locking, since they also change HEAD.\n\nAnd that's going to be a lot easier if we only do a cmp-xchg and fail at\nthe end (this concurrent change and \"remote repo\" usage is really quite\nwrong, since it basically mean that you consider somebody's development\ntree to be \"public\", and that's not what git is all about, but whatever..)\n\nBut this is not a huge issue. The most important thing is to make sure\nthat the new HEAD is written last, regardless, so that at least local\nreaders (including things like \"fsck\" that can take a _loong_ time) always\nsee consistent state.\n\n> I would assume the biggest problem for most people is how they can push\n> through a firewall. From that point of view it would make sense to do\n> the push as a cgi script rather than something over ssh. And with a\n> cgi script you can of course optimize the locking and use tmp files\n> before renaming which are a bit hard to do with rsync.\n\nMe personally, I want ssh as a major option. There are tons of machines \n(every single of my own ones) that I use that don't let anything but ssh \nthrough.\n\nBut having alternatives is good. But ssh should be the first and primary \none, since it also means that there can't be any new security issues (ie \nyou won't be opening up any new holes by installing git on the remote \nmachine).\n\n\t\tLinus\n"},{"id":"4429","messageId":"20050602175419.GD21363@atomide.com","threadId":"779","inReplyTo":"Pine.LNX.4.58.0506020741480.1876@ppc970.osdl.org","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Tony Lindgren","fromEmail":"tony@atomide.com","sentAt":"2005-06-02T17:54:19Z","receivedAt":"2005-06-02T17:54:19Z","isPatch":false,"sender":{"key":"tony@atomide.com","avatar":"https://gravatar.com/avatar/d536761d645f787e43f0491b2c7c003f9e3c1041a5458caafce7e5b5618a6530?d=mp&s=160"},"body":"* Linus Torvalds <torvalds@osdl.org> [050602 07:50]:\n> \n> \n> On Thu, 2 Jun 2005, Tony Lindgren wrote:\n> > \n> > I don't think locking for the duration of the push really is a problem.\n> > It is unlikely that there would be so many people pushing that it would\n> > cause inconvenience... Of course it would be nice to optimize it if\n> > possible.\n> \n> I don't think the locking is a _huge_ issue - the only real problem I had\n> with locking with BK was that readers could block a writer (ie I couldn't\n> push to my public tree if there were people downloading from it), and git\n> doesn't have that problem. I only push to trees that are my private ones\n> anyway.\n> \n> The real reason I'd prefer to not do locking is that _if_ the remote tree\n> is actually more than just a CVS-like \"public repository\", ie if somebody\n> actually does _development_ in the remote tree (hey, it may be crazy, but\n> git makes this usage pattern possible), then we should eventually plan on\n> having all of the regular \"git-commit-script\" and \"git-pull-script\" etc\n> _also_ do locking, since they also change HEAD.\n\nOr perhaps a more likely scenario would be automated pulls from other trees\nrunning as cronjobs on the remote server.\n\n> And that's going to be a lot easier if we only do a cmp-xchg and fail at\n> the end (this concurrent change and \"remote repo\" usage is really quite\n> wrong, since it basically mean that you consider somebody's development\n> tree to be \"public\", and that's not what git is all about, but whatever..)\n> \n> But this is not a huge issue. The most important thing is to make sure\n> that the new HEAD is written last, regardless, so that at least local\n> readers (including things like \"fsck\" that can take a _loong_ time) always\n> see consistent state.\n> \n> > I would assume the biggest problem for most people is how they can push\n> > through a firewall. From that point of view it would make sense to do\n> > the push as a cgi script rather than something over ssh. And with a\n> > cgi script you can of course optimize the locking and use tmp files\n> > before renaming which are a bit hard to do with rsync.\n> \n> Me personally, I want ssh as a major option. There are tons of machines \n> (every single of my own ones) that I use that don't let anything but ssh \n> through.\n\nYeah I prefer ssh too in general. Many corporate firewalls don't allow\nssh through though.\n\n> But having alternatives is good. But ssh should be the first and primary \n> one, since it also means that there can't be any new security issues (ie \n> you won't be opening up any new holes by installing git on the remote \n> machine).\n\nYeah. Doing it as cgi also has some issues getting the file\npermissions right so the repo would still work for ssh users too...\n\nIs anybody planning to work on this? I'm pretty much out of\ntime and will probably be happy with the rsync script for now.\n\nTony\n"},{"id":"4436","messageId":"Pine.LNX.4.21.0506021452030.30848-100000@iabervon.org","threadId":"779","inReplyTo":"Pine.LNX.4.58.0506020741480.1876@ppc970.osdl.org","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Daniel Barkalow","fromEmail":"barkalow@iabervon.org","sentAt":"2005-06-02T19:12:31Z","receivedAt":"2005-06-02T19:12:31Z","isPatch":false,"sender":{"key":"barkalow@iabervon.org","avatar":"https://avatars.githubusercontent.com/u/55364219?v=4"},"body":"On Thu, 2 Jun 2005, Linus Torvalds wrote:\n\n> The real reason I'd prefer to not do locking is that _if_ the remote tree\n> is actually more than just a CVS-like \"public repository\", ie if somebody\n> actually does _development_ in the remote tree (hey, it may be crazy, but\n> git makes this usage pattern possible), then we should eventually plan on\n> having all of the regular \"git-commit-script\" and \"git-pull-script\" etc\n> _also_ do locking, since they also change HEAD.\n\nIt might be okay to have a single tree to which is applied patches from\nemails sent by non-maintainers, pulls from non-maintainers with accessible\nrepositories, and pushes from home repositories of maintainers. It makes\nsense to deal well with different ways to get refined commits into the\npublic repository. Of course, you probably don't want to carry out merges\nthere, so the cases where pull or commit loses the race are quite\ndifferent; you want to throw it back to the author for rebasing.\n\n\t-Daniel\n*This .sig left intentionally blank*\n\n"},{"id":"4437","messageId":"429F5AC9.5080107@gmail.com","threadId":"779","inReplyTo":"20050531190005.GE18723@atomide.com","subject":"Re: [SCRIPT] cg-rpush & locking","fromName":"Dan Holmsand","fromEmail":"holmsand@gmail.com","sentAt":"2005-06-02T19:15:21Z","receivedAt":"2005-06-02T19:15:21Z","isPatch":false,"sender":{"key":"holmsand@gmail.com","avatar":"https://gravatar.com/avatar/5c722084bafd85e754a02efad01fe69107eb6f393253c49232c5c9f7faa974df?d=mp&s=160"},"body":"Tony Lindgren wrote:\n> Anybody have any better ideas for locking that also works with\n> rsync?\n\nSince this seems to be Push Day on the git list, here's my feeble \nattempt at the same thing (attached cg-push script).\n\nI do stuff in this order:\n\n     1. read remote HEAD\n     2. check that a merge from local HEAD to remote would be\n        fast-forward, otherwise tell people to pull and merge.\n     3. push objects using --ignore-existing (or equivalent)\n     4. write lock file with --ignore-existing. The lock file\n        contains, in particular, the HEAD to be written.\n     5. read remote HEAD (again) and the lock file. Bail if HEAD\n        changed since the first read, or if the lock file isn't\n        the one we attempted to write.\n     6. write remote HEAD and delete lock file in using rsync's\n        --delete-after\n\nThis should always be safe, since rsync (as I understand the man page) \nalways writes to temp files, and then renames into place. Checking for \nfast-forward-mergability assures that other peoples changes don't get lost.\n\ncg-push determines uri and foreign branch name using the same rules as \ncg-pull, which is real nice as it allows you to do:\n\n$ cg-clone me@myserver.example.com:git-repos/myrepo.git mystuff\n$ cd mystuff\n# write some stuff\n$ cg-commit\n$ cg-push\n\nand later\n\n$ cg-update\n# write more stuff\n$ cg-commit\n$ cg-push\n\nand even later\n\n$ cg-branch-add pub me@myserver.example.com:public-repos/myrepo.git\n$ cg-push pub\n\nSo, you can work pretty much exactly as you would do in CVS or svn, if \nyou're so inclined, safely sharing a common repository among many users.\n\nWhich is kinda neat, if I may say so myself...\n\n/dan\n\n\n#! /usr/bin/env bash\n#\n# Push changes to a remote git repository\n#\n# Copyright (c) Dan Holmsand, 2005.\n#\n# Based on cg-pull,\n# Copyright (c) Petr Baudis, 2005.\n#\n# Takes the branch name as an argument, defaulting to \"origin\" (see\n# `cg-branch-add` for some description of branch names).\n#\n# Takes one optional option: --force, that makes cg-push write objects\n# regardless of lock-files and remote state. Use with care...\n#\n# cg-push supports two types of location specifiers:\n#\n# 1. local paths - simple directory names of git repositories\n# 2. rsync - using the \"[user@]machine:some/path\" syntax\n#\n# Typical use would look something like this:\n#\n#\t# clone a remote branch:\n#\tcg-clone me@myserver.example.com:repo.git myrepo\n#\tcd myrepo\n#\n#\t# make some changes, and then do:\n#\tcg-commit\n#\tcg-push\n#\t\n# cg-push is safe to use even if multiple users concurrently push\n# to the same repository. Here's how this works:\n#\n# First, cg-push checks that the local repository is fully merged with\n# the remote one (this will always be the case if there is only one\n# user). Otherwise, you need to cg-update from the remote repository\n# before you can cg-push to it.\n#\n# Then, cg-push writes all the object files that are missing in the\n# remote repository.\n#\n# To finish, cg-push writes a lock file to the remote site (ignoring\n# any preexisting lock file), checks that our lock file actually got \n# written, checks that the remote head is still the same (i.e. that \n# the remote site hasn't been updated while we were copying objects), \n# writes the new remote head and removes the lock.\n#\n# The head of the local repository is also updated in the process (as if\n# the remote branch had been cg-pull'ed).\n#\n# cg-push requires that there already is a repository in place at the\n# remote location, but it actually only checks that it has a \n# \"refs/heads\" subdirectory. So, creating a remote repo ready for\n# cg-pushing is as easy \"mkdir -p repo.git/heads/refs\" at the remote\n# location.\n\n# TODO: Write tags as well.\n\n. ${COGITO_LIB}cg-Xlib\n\nforce=\nif [ \"$1\" = --force ]; then\n\tforce=1; shift\nfi\n\nname=$1\n[ \"$name\" ] || { [ -s $_git/refs/heads/origin ] && name=origin; }\n[ \"$name\" ] || die \"what to push to?\"\nuri=$(cat \"$_git/branches/$name\" 2>/dev/null) || die \"unknown branch: $name\"\n\nrembranch=master\nif echo \"$uri\" | grep -q '#'; then\n\trembranch=$(echo $uri | cut -d '#' -f 2)\n\turi=$(echo $uri | cut -d '#' -f 1)\nfi\n\ncase $uri in\n\t*:*)\n\treadhead=rsync_readhead\n\twriteobjects=rsync_writeobjects\n\twritehead=rsync_writehead\n\tlock=rsync_lock\n\t;;\n\t*)\n\tif [ -d \"$uri\" ]; then\n\t\t[ -d \"$uri/.git\" ] && uri=$uri/.git\n\t\treadhead=local_readhead\n\t\twriteobjects=local_writeobjects\n\t\twritehead=local_writehead\n\t\tlock=local_lock\n\telse\n\t\tdie \"Don't know how to push to $uri\"\n\tfi\n\t;;\nesac\n\ntmpd=$(mktemp -d -t cgpush.XXXXXX) || exit 1\ntrap \"rm -rf $tmpd\" SIGTERM EXIT\n\ncid=$(commit-id) || exit 1\nlock_msg=\"locked by $USER@$HOSTNAME on $(date) for writing $cid\"\nunset locked remhead\n\nrsync_readhead() {\n\trm -f \"$tmpd/*\" || return 1\n\trsync $RSYNC_FLAGS --include=\"$rembranch\" --include=\"$rembranch.lock\" \\\n\t\t--exclude='*' -r \"$uri/refs/heads/\" \"$tmpd/\" >&2 || \n\t\tdie \"Fetching heads from $uri failed. Aborting.\"\n\tif [ \"$locked\" ]; then\n\t\t[ -s \"$tmpd/$rembranch.lock\" ] ||\n\t\tdie \"Couldn't acquire lock. Aborting.\"\n\n\t\tlocal rem_lock_msg=$(cat \"$tmpd/$rembranch.lock\")\n\t\t[ \"$lock_msg\" = \"$rem_lock_msg\" ] ||\n\t\tdie \"Remote is locked ($rem_lock_msg).\"\n\tfi\n\t[ ! -e \"$tmpd/$rembranch\" ] || cat \"$tmpd/$rembranch\"\n}\n\nrsync_writeobjects() {\n\t[ -d \"$_git/objects/\" ] || die \"no objects to copy\"\n\trsync $RSYNC_FLAGS -vr --ignore-existing --whole-file \\\n\t\t\"$_git/objects/\" \"$uri/objects/\" \n}\n\nrsync_lock() {\n\techo \"$lock_msg\" > $tmpd/new_head_lock_file || return 1\n\trsync $RSYNC_FLAGS --ignore-existing --whole-file \\\n\t\t$tmpd/new_head_lock_file \"$uri/refs/heads/$rembranch.lock\"\n}\n\nrsync_writehead() {\n\tlocal heads=$tmpd/newhead\n\tmkdir $heads && echo \"$1\" > \"$heads/$rembranch\" || return 1\n\trsync $RSYNC_FLAGS --include=\"$rembranch\" --include=\"$rembranch.lock\" \\\n\t\t--exclude='*' --delete-after -r $heads/ \"$uri/refs/heads/\" \n}\n\nlocal_readhead() {\n\tlocal lheads=$uri/refs/heads\n\t[ -d \"$lheads\" ] || die \"no remote heads found at $uri\"\n\t[ ! -e \"$lheads/$rembranch\" ] || cat \"$lheads/$rembranch\" \n}\n\nlocal_writeobjects() {\n\t[ -d \"$_git/objects/\" ] || die \"no objects to copy\"\n\t[ -d \"$uri/objects\" ] || \n\t\tGIT_DIR=$uri GIT_OBJECT_DIRECTORY=$uri/objects git-init-db ||\n\t\tdie \"git-init-db failed\"\n        # Note: We could use git-local-pull here, but this is safer\n\t# (git-*-pull don't react well to failures or kills), and\n\t# has the same semantics as rsync pushing.\n\tlocal dest=$(cd \"$uri/objects\" && pwd) || exit 1\n\t( cd \"$_git/objects\" && find -type f | while read f; do\n\t\t[ -f \"$dest/$f\" ] && continue\n\t\tln \"$f\" \"$dest/$f\" 2>/dev/null || \n\t\tcp \"$f\" \"$dest/$f\" || exit 1\n\tdone ) \n}\n\nlocal_lock() {\n\t([ \"$force\" ] || set -C \n\techo \"$lockmsg\" > \"$uri/refs/heads/$rembranch.lock\") 2>/dev/null\n}\n\nlocal_writehead() {\n\tlocal head=$uri/refs/heads/$rembranch\n\techo \"$1\" > \"$head.new\" && mv \"$head.new\" \"$head\" &&\n\trm \"$head.lock\"\n}\n\necho \"Checking remote repository\"\nremhead=$($readhead) || exit 1\n[ \"$remhead\" ] || echo \"Creating new branch\"\n\nif [ \"$remhead\" -a -z \"$force\" ]; then\n\tif [ \"$remhead\" = \"$cid\" ]; then\n\t\techo \"Remote branch \\`$name' is already pushed\" \n\t\texit 0\n\tfi\n\tgit-cat-file commit \"$remhead\" &> /dev/null ||\n\tdie \"You need to pull from $name first. Aborting.\"\n\n\tbase=$(git-merge-base \"$remhead\" \"$cid\") && [ \"$base\" ] ||\n\tdie \"You need to merge $name. Aborting.\"\n\n\tif [ \"$base\" = \"$cid\" ]; then\n\t\techo \"No changes to push\"; exit 0\n\tfi\n\n\t[ \"$base\" = \"$remhead\" ] || \n\tdie \"You need to merge $name first. Aborting.\" \nfi\n\necho \"Writing objects\"\n$writeobjects || die \"Failed to write objects. Aborting.\"\n\necho\necho \"Writing new head\"\n$lock || die \"Couldn't acquire lock on remote. Aborting.\"\n\nif [ ! \"$force\" ]; then\n\tlocked=1\n\tremhead2=$($readhead) || die \"Aborting.\"\n\t[ \"$remhead\" = \"$remhead2\" ] || \n\t\tdie \"Remote head changed during copy. Aborting.\"\nfi\n\n$writehead \"$cid\" || die \"WARNING: Error writing remote head. Aborting.\"\necho \"Push to $name succeeded\"\n\necho \"$cid\" > \"$_git/refs/heads/$name\"\necho \"Updated local head for $name to $cid\"\n\n"}]}