summaryrefslogtreecommitdiff
path: root/django/utils/encoding.py
diff options
context:
space:
mode:
authorMalcolm Tredinnick <malcolm.tredinnick@gmail.com>2007-04-09 10:33:57 +0000
committerMalcolm Tredinnick <malcolm.tredinnick@gmail.com>2007-04-09 10:33:57 +0000
commitb493b7e3cf09eb0df5c460e8ca7b6a934e40e43c (patch)
treea7884ca5c4ef65902d934e517a727fbe346cd56c /django/utils/encoding.py
parent232b7ac519255b95ac0b121c59fb475cdbabbc04 (diff)
unicode: Converted the template output and database I/O interfaces to
understand unicode strings. All tests pass (except for one commented out with "XFAIL"), but untested with database servers using non-UTF8, non-ASCII on the server. git-svn-id: http://code.djangoproject.com/svn/django/branches/unicode@4971 bcc190cf-cafb-0310-a4f2-bffc1f526a37
Diffstat (limited to 'django/utils/encoding.py')
-rw-r--r--django/utils/encoding.py51
1 files changed, 39 insertions, 12 deletions
diff --git a/django/utils/encoding.py b/django/utils/encoding.py
index 4774fb0d26..eb6b63660e 100644
--- a/django/utils/encoding.py
+++ b/django/utils/encoding.py
@@ -1,25 +1,50 @@
+import types
from django.conf import settings
from django.utils.functional import Promise
-def smart_unicode(s):
- if isinstance(s, Promise):
- # The input is the result of a gettext_lazy() call, or similar. It will
- # already be encoded in DEFAULT_CHARSET on evaluation and we don't want
- # to evaluate it until render time.
- # FIXME: This isn't totally consistent, because it eventually returns a
- # bytestring rather than a unicode object. It works wherever we use
- # smart_unicode() at the moment. Fixing this requires work in the
- # i18n internals.
- return s
+def smart_unicode(s, encoding='utf-8'):
+ """
+ Returns a unicode object representing 's'. Treats bytestrings using the
+ 'encoding' codec.
+ """
+ #if isinstance(s, Promise):
+ # # The input is the result of a gettext_lazy() call, or similar. It will
+ # # already be encoded in DEFAULT_CHARSET on evaluation and we don't want
+ # # to evaluate it until render time.
+ # # FIXME: This isn't totally consistent, because it eventually returns a
+ # # bytestring rather than a unicode object. It works wherever we use
+ # # smart_unicode() at the moment. Fixing this requires work in the
+ # # i18n internals.
+ # return s
if not isinstance(s, basestring,):
if hasattr(s, '__unicode__'):
s = unicode(s)
else:
- s = unicode(str(s), settings.DEFAULT_CHARSET)
+ s = unicode(str(s), encoding)
elif not isinstance(s, unicode):
- s = unicode(s, settings.DEFAULT_CHARSET)
+ s = unicode(s, encoding)
return s
+def smart_str(s, encoding='utf-8', strings_only=False):
+ """
+ Returns a bytestring version of 's', encoded as specified in 'encoding'.
+
+ If strings_only is True, don't convert (some) non-string-like objects.
+ """
+ if strings_only and isinstance(s, (types.NoneType, int)):
+ return s
+ if not isinstance(s, basestring):
+ try:
+ return str(s)
+ except UnicodeEncodeError:
+ return unicode(s).encode(encoding)
+ elif isinstance(s, unicode):
+ return s.encode(encoding)
+ elif s and encoding != 'utf-8':
+ return s.decode('utf-8').encode(encoding)
+ else:
+ return s
+
class StrAndUnicode(object):
"""
A class whose __str__ returns its __unicode__ as a bytestring
@@ -28,5 +53,7 @@ class StrAndUnicode(object):
Useful as a mix-in.
"""
def __str__(self):
+ # XXX: (Malcolm) Correct encoding? Be variable and use UTF-8 as
+ # default?
return self.__unicode__().encode(settings.DEFAULT_CHARSET)