diff options
| author | Malcolm Tredinnick <malcolm.tredinnick@gmail.com> | 2007-04-09 10:33:57 +0000 |
|---|---|---|
| committer | Malcolm Tredinnick <malcolm.tredinnick@gmail.com> | 2007-04-09 10:33:57 +0000 |
| commit | b493b7e3cf09eb0df5c460e8ca7b6a934e40e43c (patch) | |
| tree | a7884ca5c4ef65902d934e517a727fbe346cd56c /django/utils/encoding.py | |
| parent | 232b7ac519255b95ac0b121c59fb475cdbabbc04 (diff) | |
unicode: Converted the template output and database I/O interfaces to
understand unicode strings. All tests pass (except for one commented out with
"XFAIL"), but untested with database servers using non-UTF8, non-ASCII on the
server.
git-svn-id: http://code.djangoproject.com/svn/django/branches/unicode@4971 bcc190cf-cafb-0310-a4f2-bffc1f526a37
Diffstat (limited to 'django/utils/encoding.py')
| -rw-r--r-- | django/utils/encoding.py | 51 |
1 files changed, 39 insertions, 12 deletions
diff --git a/django/utils/encoding.py b/django/utils/encoding.py index 4774fb0d26..eb6b63660e 100644 --- a/django/utils/encoding.py +++ b/django/utils/encoding.py @@ -1,25 +1,50 @@ +import types from django.conf import settings from django.utils.functional import Promise -def smart_unicode(s): - if isinstance(s, Promise): - # The input is the result of a gettext_lazy() call, or similar. It will - # already be encoded in DEFAULT_CHARSET on evaluation and we don't want - # to evaluate it until render time. - # FIXME: This isn't totally consistent, because it eventually returns a - # bytestring rather than a unicode object. It works wherever we use - # smart_unicode() at the moment. Fixing this requires work in the - # i18n internals. - return s +def smart_unicode(s, encoding='utf-8'): + """ + Returns a unicode object representing 's'. Treats bytestrings using the + 'encoding' codec. + """ + #if isinstance(s, Promise): + # # The input is the result of a gettext_lazy() call, or similar. It will + # # already be encoded in DEFAULT_CHARSET on evaluation and we don't want + # # to evaluate it until render time. + # # FIXME: This isn't totally consistent, because it eventually returns a + # # bytestring rather than a unicode object. It works wherever we use + # # smart_unicode() at the moment. Fixing this requires work in the + # # i18n internals. + # return s if not isinstance(s, basestring,): if hasattr(s, '__unicode__'): s = unicode(s) else: - s = unicode(str(s), settings.DEFAULT_CHARSET) + s = unicode(str(s), encoding) elif not isinstance(s, unicode): - s = unicode(s, settings.DEFAULT_CHARSET) + s = unicode(s, encoding) return s +def smart_str(s, encoding='utf-8', strings_only=False): + """ + Returns a bytestring version of 's', encoded as specified in 'encoding'. + + If strings_only is True, don't convert (some) non-string-like objects. + """ + if strings_only and isinstance(s, (types.NoneType, int)): + return s + if not isinstance(s, basestring): + try: + return str(s) + except UnicodeEncodeError: + return unicode(s).encode(encoding) + elif isinstance(s, unicode): + return s.encode(encoding) + elif s and encoding != 'utf-8': + return s.decode('utf-8').encode(encoding) + else: + return s + class StrAndUnicode(object): """ A class whose __str__ returns its __unicode__ as a bytestring @@ -28,5 +53,7 @@ class StrAndUnicode(object): Useful as a mix-in. """ def __str__(self): + # XXX: (Malcolm) Correct encoding? Be variable and use UTF-8 as + # default? return self.__unicode__().encode(settings.DEFAULT_CHARSET) |
