|
9 | 9 | 'add_codec', |
10 | 10 | ] |
11 | 11 |
|
| 12 | +import codecs |
12 | 13 | from functools import partial |
13 | 14 |
|
14 | 15 | import email.base64mime |
|
58 | 59 | 'shift_jis': (BASE64, None, 'iso-2022-jp'), |
59 | 60 | 'iso-2022-jp': (BASE64, None, None), |
60 | 61 | 'koi8-r': (BASE64, BASE64, None), |
61 | | - 'utf-8': (SHORTEST, BASE64, 'utf-8'), |
62 | 62 | } |
63 | 63 |
|
64 | | -# Aliases for other commonly-used names for character sets. Map |
65 | | -# them to the real ones used in email. |
| 64 | +# Map Python codec names to their corresponding MIME/IANA names. |
66 | 65 | ALIASES = { |
67 | | - 'latin_1': 'iso-8859-1', |
68 | | - 'latin-1': 'iso-8859-1', |
69 | | - 'latin_2': 'iso-8859-2', |
70 | | - 'latin-2': 'iso-8859-2', |
71 | | - 'latin_3': 'iso-8859-3', |
72 | | - 'latin-3': 'iso-8859-3', |
73 | | - 'latin_4': 'iso-8859-4', |
74 | | - 'latin-4': 'iso-8859-4', |
75 | | - 'latin_5': 'iso-8859-9', |
76 | | - 'latin-5': 'iso-8859-9', |
77 | | - 'latin_6': 'iso-8859-10', |
78 | | - 'latin-6': 'iso-8859-10', |
79 | | - 'latin_7': 'iso-8859-13', |
80 | | - 'latin-7': 'iso-8859-13', |
81 | | - 'latin_8': 'iso-8859-14', |
82 | | - 'latin-8': 'iso-8859-14', |
83 | | - 'latin_9': 'iso-8859-15', |
84 | | - 'latin-9': 'iso-8859-15', |
85 | | - 'latin_10':'iso-8859-16', |
86 | | - 'latin-10':'iso-8859-16', |
87 | | - 'cp949': 'ks_c_5601-1987', |
88 | | - 'euc_jp': 'euc-jp', |
89 | | - 'euc_kr': 'euc-kr', |
90 | | - 'ascii': 'us-ascii', |
91 | | - } |
| 66 | + 'ascii': 'us-ascii', |
| 67 | + 'big5hkscs': 'big5-hkscs', |
| 68 | + 'cp037': 'ibm037', |
| 69 | + 'cp1026': 'ibm1026', |
| 70 | + 'cp1140': 'ibm01140', |
| 71 | + 'cp1250': 'windows-1250', |
| 72 | + 'cp1251': 'windows-1251', |
| 73 | + 'cp1252': 'windows-1252', |
| 74 | + 'cp1253': 'windows-1253', |
| 75 | + 'cp1254': 'windows-1254', |
| 76 | + 'cp1255': 'windows-1255', |
| 77 | + 'cp1256': 'windows-1256', |
| 78 | + 'cp1257': 'windows-1257', |
| 79 | + 'cp1258': 'windows-1258', |
| 80 | + 'cp273': 'ibm273', |
| 81 | + 'cp424': 'ibm424', |
| 82 | + 'cp437': 'ibm437', |
| 83 | + 'cp500': 'ibm500', |
| 84 | + 'cp775': 'ibm775', |
| 85 | + 'cp850': 'ibm850', |
| 86 | + 'cp852': 'ibm852', |
| 87 | + 'cp855': 'ibm855', |
| 88 | + 'cp857': 'ibm857', |
| 89 | + 'cp858': 'ibm00858', |
| 90 | + 'cp860': 'ibm860', |
| 91 | + 'cp861': 'ibm861', |
| 92 | + 'cp862': 'ibm862', |
| 93 | + 'cp863': 'ibm863', |
| 94 | + 'cp864': 'ibm864', |
| 95 | + 'cp865': 'ibm865', |
| 96 | + 'cp866': 'ibm866', |
| 97 | + 'cp869': 'ibm869', |
| 98 | + 'cp874': 'windows-874', |
| 99 | + 'euc_jp': 'euc-jp', |
| 100 | + 'euc_kr': 'euc-kr', |
| 101 | + 'hz': 'hz-gb-2312', |
| 102 | + 'iso2022_jp': 'iso-2022-jp', |
| 103 | + 'iso2022_jp_2': 'iso-2022-jp-2', |
| 104 | + 'iso2022_kr': 'iso-2022-kr', |
| 105 | + 'iso8859-1': 'iso-8859-1', |
| 106 | + 'iso8859-10': 'iso-8859-10', |
| 107 | + 'iso8859-11': 'iso-8859-11', |
| 108 | + 'iso8859-13': 'iso-8859-13', |
| 109 | + 'iso8859-14': 'iso-8859-14', |
| 110 | + 'iso8859-15': 'iso-8859-15', |
| 111 | + 'iso8859-16': 'iso-8859-16', |
| 112 | + 'iso8859-2': 'iso-8859-2', |
| 113 | + 'iso8859-3': 'iso-8859-3', |
| 114 | + 'iso8859-4': 'iso-8859-4', |
| 115 | + 'iso8859-5': 'iso-8859-5', |
| 116 | + 'iso8859-6': 'iso-8859-6', |
| 117 | + 'iso8859-7': 'iso-8859-7', |
| 118 | + 'iso8859-8': 'iso-8859-8-i', |
| 119 | + 'iso8859-9': 'iso-8859-9', |
| 120 | + 'kz1048': 'kz-1048', |
| 121 | + 'mac-roman': 'macintosh', |
| 122 | + |
| 123 | + # CP949 is not registered in IANA. KS_C_5601-1987 is not the same, |
| 124 | + # but the closest registered option. |
| 125 | + 'cp949': 'ks_c_5601-1987', |
| 126 | +} |
92 | 127 |
|
93 | 128 |
|
94 | 129 | # Map charsets to their Unicode codec strings. |
@@ -215,7 +250,18 @@ def __init__(self, input_charset=DEFAULT_CHARSET): |
215 | 250 | raise errors.CharsetError(input_charset) |
216 | 251 | input_charset = input_charset.lower() |
217 | 252 | # Set the input charset after filtering through the aliases |
218 | | - self.input_charset = ALIASES.get(input_charset, input_charset) |
| 253 | + # For backward compatibility, try ALIASES first to let the user |
| 254 | + # override it. |
| 255 | + if input_charset in ALIASES: |
| 256 | + input_charset = ALIASES[input_charset] |
| 257 | + else: |
| 258 | + try: |
| 259 | + input_codec = codecs.lookup(input_charset).name |
| 260 | + except LookupError: |
| 261 | + pass |
| 262 | + else: |
| 263 | + input_charset = ALIASES.get(input_codec, input_codec) |
| 264 | + self.input_charset = input_charset |
219 | 265 | # We can try to guess which encoding and conversion to use by the |
220 | 266 | # charset_map dictionary. Try that first, but let the user override |
221 | 267 | # it. |
|
0 commit comments