diff --git a/dump_html.py b/dump_html.py
index 65adc34..d164f8a 100755
--- a/dump_html.py
+++ b/dump_html.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python2
# -*- coding: UTF-8 -*-
# File: dump_html.py
-# Date: Tue Dec 23 00:01:20 2014 +0800
+# Date: Fri Jan 09 22:16:26 2015 +0800
# Author: Yuxin Wu
import sys
@@ -14,12 +14,13 @@ from lib.res import Resource
from lib.render import HTMLRender
db_file = sys.argv[1]
-res = Resource(sys.argv[2])
+resource_dir = sys.argv[2]
name = ensure_unicode(sys.argv[3])
output_file = sys.argv[4]
parser = WeChatDBParser(db_file)
msgs = parser.msgs_by_talker[name]
+res = Resource(resource_dir)
render = HTMLRender(parser, res)
htmls = render.render_msgs(msgs)
diff --git a/lib/__init__.py b/lib/__init__.py
index c7a7888..282de17 100644
--- a/lib/__init__.py
+++ b/lib/__init__.py
@@ -1,6 +1,30 @@
+#!/usr/bin/env python2
+# -*- coding: UTF-8 -*-
import logging
-logging.basicConfig(
- format='\033[1;32m[%(asctime)s %(lineno)d@%(filename)s:%(name)s]\033[0m'
- ' %(message)s',
- datefmt='%H:%M:%S', level=logging.INFO)
+logger = logging.getLogger(__name__)
+logger.setLevel(logging.INFO)
+
+FMT = '{}[%(asctime)s %(lineno)d@%(filename)s:%(name)s]\033[0m %(message)s'
+
+class LogLevelFilter(object):
+ def __init__(self, level):
+ self._level = level
+ def filter(self, logRecord):
+ return logRecord.levelno <= self._level
+
+def set_level_color(lvl, color):
+ handler = logging.StreamHandler()
+ handler.setLevel(lvl)
+ handler.addFilter(LogLevelFilter(lvl))
+ handler.setFormatter(logging.Formatter(FMT.format(color), '%H:%M:%S'))
+ logger.addHandler(handler)
+
+set_level_color(logging.INFO, '\033[1;32m')
+set_level_color(logging.WARN, '\033[1;31m')
+set_level_color(logging.ERROR, '\033[1;31m')
+
+if __name__ == '__main__':
+ logger.info("info")
+ logger.warn("warn")
+
diff --git a/lib/avatar.py b/lib/avatar.py
index d8b8ad8..3d059ec 100644
--- a/lib/avatar.py
+++ b/lib/avatar.py
@@ -1,14 +1,13 @@
#!/usr/bin/env python2
# -*- coding: UTF-8 -*-
# File: avatar.py
-# Date: Wed Dec 17 00:08:06 2014 +0800
+# Date: Fri Jan 09 22:21:55 2015 +0800
# Author: Yuxin Wu
import os
-import hashlib
import numpy as np
-from .utils import ensure_bin_str
+from .utils import ensure_bin_str, md5
class AvatarReader(object):
def __init__(self, res_dir):
@@ -16,16 +15,10 @@ class AvatarReader(object):
assert os.path.isdir(self.avt_dir), \
"No such directory {}".format(self.avt_dir)
- @staticmethod
- def get_filename(username):
- m = hashlib.md5()
- m.update(username)
- return m.hexdigest()
-
def get_avatar(self, username):
""" username: `username` field in db.rcontact"""
username = ensure_bin_str(username)
- filename = AvatarReader.get_filename(username)
+ filename = md5(username)
dir1, dir2 = filename[:2], filename[2:4]
filename = os.path.join(self.avt_dir, dir1, dir2,
"user_{}.png.bm".format(filename))
diff --git a/lib/msg.py b/lib/msg.py
index dc194de..0940619 100644
--- a/lib/msg.py
+++ b/lib/msg.py
@@ -1,15 +1,8 @@
#!/usr/bin/env python2
# -*- coding: UTF-8 -*-
# File: msg.py
-# Date: Wed Jan 07 23:59:45 2015 +0800
+# Date: Fri Jan 09 22:14:53 2015 +0800
# Author: Yuxin Wu
-
-import re
-from datetime import datetime
-from pyquery import PyQuery
-
-from .utils import ensure_bin_str, ensure_unicode
-
TYPE_MSG = 1
TYPE_IMG = 3
TYPE_SPEAK = 34
@@ -25,6 +18,17 @@ TYPE_CUSTOM_EMOJI = 1048625
TYPE_LOCATION_SHARING = -1879048186
TYPE_APP_MSG = 16777265
+_KNOWN_TYPES = [eval(k) for k in dir() if k.startswith('TYPE_')]
+
+import re
+from datetime import datetime
+from pyquery import PyQuery
+import logging
+logger = logging.getLogger(__name__)
+
+from .utils import ensure_bin_str, ensure_unicode
+
+
class WeChatMsg(object):
FIELDS = ["msgSvrId","type","isSend","createTime","talker","content","imgPath"]
@@ -39,6 +43,10 @@ class WeChatMsg(object):
assert len(row) == len(WeChatMsg.FIELDS)
for f, v in zip(WeChatMsg.FIELDS, row):
setattr(self, f, v)
+ if self.type not in _KNOWN_TYPES:
+ logger.warn("Unhandled message type: {}".format(self.type))
+ # only to supress repeated warning:
+ _KNOWN_TYPES.append(self.type)
self.createTime = datetime.fromtimestamp(self.createTime / 1000)
self.talker_name = None
if self.content:
diff --git a/lib/res.py b/lib/res.py
index 372f085..98af729 100644
--- a/lib/res.py
+++ b/lib/res.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python2
# -*- coding: UTF-8 -*-
# File: res.py
-# Date: Fri Jan 09 11:52:53 2015 +0800
+# Date: Fri Jan 09 22:28:54 2015 +0800
# Author: Yuxin Wu
import glob
@@ -12,14 +12,14 @@ import Image
import cStringIO
import base64
import logging
+logger = logging.getLogger(__name__)
import imghdr
from multiprocessing import Pool
-logger = logging.getLogger(__name__)
import pysox
from .avatar import AvatarReader
-from .utils import timing
+from .utils import timing, md5
LIB_PATH = os.path.dirname(os.path.abspath(__file__))
INTERNAL_EMOJI_DIR = os.path.join(LIB_PATH, 'static', 'internal_emoji')
@@ -30,8 +30,7 @@ JPEG_QUALITY = 50
def do_get_voice_mp3(amr_fpath):
""" return base64 string, and voice duration"""
- assert amr_fpath.endswith('.amr')
- assert os.path.isfile(amr_fpath), amr_fpath
+ if not amr_fpath: return "", 0
mp3_file = os.path.join('/tmp',
os.path.basename(amr_fpath)[:-4] + '.mp3')
@@ -55,37 +54,32 @@ class Resource(object):
self.img_dir = os.path.join(res_dir, IMG_DIRNAME)
self.emoji_dir = os.path.join(res_dir, EMOJI_DIRNAME)
self.avt_reader = AvatarReader(self.res_dir)
- self.init()
- def init(self):
- """ load some index in memory"""
- self.speak_data = {}
- for root, subdirs, files in os.walk(
- os.path.join(self.res_dir, VOICE_DIRNAME)):
- if subdirs:
- continue
- for f in files:
- if not f.endswith('amr'):
- continue
- full_path = os.path.join(root, f)
- key = f[4:-4] # msg_xxxxx.amr
- assert len(key) == 26, \
- "Error interpreting the protocol, this is potentially a bug!"
- self.speak_data[key] = full_path
+ def get_voice_filename(self, imgpath):
+ fname = md5(imgpath)
+ dir1, dir2 = fname[:2], fname[2:4]
+ ret = os.path.join(self.res_dir, VOICE_DIRNAME, dir1, dir2,
+ 'msg_{}.amr'.format(imgpath))
+ if not os.path.isfile(ret):
+ logger.error("Voice file not found for {}".format(imgpath))
+ return ""
+ return ret
@timing(total=True)
def get_voice_mp3(self, imgpath):
+ """ return mp3 and duration, or empty string and 0 on failure"""
idx = self.voice_cache_idx.get(imgpath)
if idx is None:
- return do_get_voice_mp3(self.speak_data[imgpath])
+ return do_get_voice_mp3(
+ self.get_voice_filename(imgpath))
return self.voice_cache[idx].get()
def cache_voice_mp3(self, voice_paths):
- """ for speed """
+ """ for speed. voice_paths: a collection of imgpath """
self.voice_cache_idx = {k: idx for idx, k in enumerate(voice_paths)}
pool = Pool(3)
self.voice_cache = [pool.apply_async(do_get_voice_mp3,
- (self.speak_data[k],)) for k in voice_paths]
+ (self.get_voice_filename(k),)) for k in voice_paths]
@staticmethod
def get_file_b64(fname):
diff --git a/lib/utils.py b/lib/utils.py
index b06c5b3..ab9733a 100644
--- a/lib/utils.py
+++ b/lib/utils.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python2
# -*- coding: UTF-8 -*-
# File: utils.py
-# Date: Thu Dec 25 16:27:39 2014 +0800
+# Date: Fri Jan 09 22:21:36 2015 +0800
# Author: Yuxin Wu
import sys
@@ -156,3 +156,9 @@ class timing(object):
return ret
return wrapper
+
+import hashlib
+def md5(s):
+ m = hashlib.md5()
+ m.update(s)
+ return m.hexdigest()