r"""Git format-patch to hg importable patch.
(Who knew this was so complicated?)
>>> process(StringIO('From 3ce1ccc06 Mon Sep 17 00:00:00 2001\nFrom: fromuser\nSubject: subject\n\nRest of patch.\nMore patch.\n'))
'# HG changeset patch\n# User fromuser\n\nsubject\n\nRest of patch.\nMore patch.\n'
>>> process(StringIO('From: fromuser\nSubject: A very long subject line. Lorem ipsum dolor sit amet, consectetur adipiscing elit. Morbi faucibus, arcu sit amet\n\nRest of patch.\nMore patch.\n'))
'# HG changeset patch\n# User fromuser\n\nA very long subject line. Lorem ipsum dolor sit amet, consectetur adipiscing elit. Morbi faucibus, arcu sit amet\n\nRest of patch.\nMore patch.\n'
>>> process(StringIO('From: f\nSubject: =?UTF-8?q?Bug=20655877=20-=20Dont=20treat=20SVG=20text=20frames=20?= =?UTF-8?q?as=20being=20positioned.=20r=3D=3F?=\n\nPatch.'))
'# HG changeset patch\n# User f\n\nBug 655877 - Dont treat SVG text frames as being positioned. r=?\n\nPatch.'
# Original author: bholley
import sys
import re
import fileinput
import email, email.parser, email.header
import math
from cStringIO import StringIO
from itertools import takewhile
def decode_header(hdr_string):
r"""Clean up weird encoding crap.
>>> clean_header('[PATCH] =?UTF-8?q?Bug=20655877=20r=3D=3F?=')
'[PATCH] Bug 655877 r=?'
rv = []
hdr = email.header.Header(hdr_string, maxlinelen=float('inf'))
for (part, encoding) in email.header.decode_header(hdr):
if encoding is None:
return ' '.join(rv)
def clean_header(hdr_string):
r"""Transform a header split over many lines into a header split only where
linebreaks are intended. This is important because hg cares about the first
line of the commit message.
Also clean up weird encoding crap.
>>> clean_header('Foo\n bar\n baz')
'Foo bar baz'
>>> clean_header('Foo\n bar\nSpam\nEggs')
'Foo bar\nSpam\nEggs'
lines = []
curline = ''
for line in decode_header(hdr_string).split('\n'):
if not line.startswith(' '):
curline = ''
curline += line
return '\n'.join(lines[1:])
def process(git_patch_file):
parser = email.parser.Parser()
msg = parser.parse(git_patch_file)
return '\n'.join(['# HG changeset patch',
'# User ' + clean_header(msg['From']),
# Decode the header into a UTF-8 string, in case it uses MIME encoded words
encoded_subject = Header(subject)
subject = ''
for (part, encoding) in decode_header(encoded_subject):
if encoding is None:
subject += part
subject += part.decode(encoding).encode('utf-8')
# Filter a version (ie, v3, v12) out of the subject.
subject = re.sub(r'\s?v\d\d?', '', subject)
# Write the header.
return ['# HG changeset patch\n', author, '\n', subject, '\n'] + list(iterlines)
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == '--test':
import doctest
# If there were no arguments, do stdin->stdout.
filelist = sys.argv[1:]
if not filelist:
lines = process(sys.stdin)
# Otherwise, we take a list of files.
for filename in filelist:
# Read the lines.
f = open(filename, 'r')
lines = process(f)
# Process.
# Write them back to the same file.
f = open(filename, 'w')