Last active
December 15, 2015 11:31
-
-
Save ikegami-yukino/68a741ef854de68871cc to your computer and use it in GitHub Desktop.
PythonでのMeCabを速くするtips
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| { | |
| "cells": [ | |
| { | |
| "cell_type": "code", | |
| "execution_count": 1, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [ | |
| { | |
| "data": { | |
| "text/plain": [ | |
| "['Python 3.5.0']" | |
| ] | |
| }, | |
| "execution_count": 1, | |
| "metadata": {}, | |
| "output_type": "execute_result" | |
| } | |
| ], | |
| "source": [ | |
| "%system python -V" | |
| ] | |
| }, | |
| { | |
| "cell_type": "markdown", | |
| "metadata": {}, | |
| "source": [ | |
| "## ドグラマグラをとってくる" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": 2, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [ | |
| { | |
| "data": { | |
| "text/plain": [ | |
| "[' % Total % Received % Xferd Average Speed Time Time Time Current',\n", | |
| " ' Dload Upload Total Spent Left Speed',\n", | |
| " '',\n", | |
| " ' 0 0 0 0 0 0 0 0 --:--:-- --:--:-- --:--:-- 0',\n", | |
| " ' 0 0 0 0 0 0 0 0 --:--:-- --:--:-- --:--:-- 0',\n", | |
| " '100 411k 100 411k 0 0 568k 0 --:--:-- --:--:-- --:--:-- 568k',\n", | |
| " 'Archive: /tmp/2093_ruby_28087.zip',\n", | |
| " 'Made with MacWinZipper™',\n", | |
| " ' inflating: /tmp/dogura_magura.txt ']" | |
| ] | |
| }, | |
| "execution_count": 2, | |
| "metadata": {}, | |
| "output_type": "execute_result" | |
| } | |
| ], | |
| "source": [ | |
| "%%system\n", | |
| "rm /tmp/2093_ruby_28087.zip /tmp/dogura_magura.txt\n", | |
| "curl -o /tmp/2093_ruby_28087.zip http://www.aozora.gr.jp/cards/000096/files/2093_ruby_28087.zip\n", | |
| "unzip -d /tmp /tmp/2093_ruby_28087.zip\n", | |
| "nkf -Sw --overwrite /tmp/dogura_magura.txt" | |
| ] | |
| }, | |
| { | |
| "cell_type": "markdown", | |
| "metadata": {}, | |
| "source": [ | |
| "## ここからベンチマーク" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": 3, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [], | |
| "source": [ | |
| "import MeCab\n", | |
| "\n", | |
| "tagger = MeCab.Tagger('-d /usr/local/lib/mecab/dic/ipadic')\n", | |
| "\n", | |
| "\n", | |
| "def preprocessing(sentence):\n", | |
| " return sentence.rstrip()" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": 4, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [ | |
| { | |
| "name": "stdout", | |
| "output_type": "stream", | |
| "text": [ | |
| "1 loops, best of 3: 667 ms per loop\n" | |
| ] | |
| } | |
| ], | |
| "source": [ | |
| "def extract_noun_by_parse(path):\n", | |
| " with open(path) as fd:\n", | |
| " nouns = []\n", | |
| " for sentence in map(preprocessing, fd):\n", | |
| " for chunk in tagger.parse(sentence).splitlines()[:-1]:\n", | |
| " (surface, feature) = chunk.split('\\t')\n", | |
| " if feature.startswith('名詞'):\n", | |
| " nouns.append(surface)\n", | |
| " return nouns\n", | |
| "\n", | |
| "\n", | |
| "%timeit extract_noun_by_parse('/tmp/dogura_magura.txt')" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": 5, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [ | |
| { | |
| "name": "stdout", | |
| "output_type": "stream", | |
| "text": [ | |
| "1 loops, best of 3: 1.62 s per loop\n" | |
| ] | |
| } | |
| ], | |
| "source": [ | |
| "def extract_noun_by_parsetonode(path):\n", | |
| " with open(path) as fd:\n", | |
| " nouns = []\n", | |
| " for sentence in map(preprocessing, fd):\n", | |
| " node = tagger.parseToNode(sentence)\n", | |
| " while node:\n", | |
| " if node.feature.startswith('名詞'):\n", | |
| " nouns.append(node.surface)\n", | |
| " node = node.next\n", | |
| " return nouns\n", | |
| "\n", | |
| "\n", | |
| "%timeit extract_noun_by_parsetonode('/tmp/dogura_magura.txt')" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": 6, | |
| "metadata": { | |
| "collapsed": false | |
| }, | |
| "outputs": [ | |
| { | |
| "name": "stdout", | |
| "output_type": "stream", | |
| "text": [ | |
| " " | |
| ] | |
| } | |
| ], | |
| "source": [ | |
| "%prun extract_noun_by_parsetonode('/tmp/dogura_magura.txt')" | |
| ] | |
| }, | |
| { | |
| "cell_type": "code", | |
| "execution_count": null, | |
| "metadata": { | |
| "collapsed": true | |
| }, | |
| "outputs": [], | |
| "source": [] | |
| } | |
| ], | |
| "metadata": { | |
| "kernelspec": { | |
| "display_name": "Python 3", | |
| "language": "python", | |
| "name": "python3" | |
| }, | |
| "language_info": { | |
| "codemirror_mode": { | |
| "name": "ipython", | |
| "version": 3 | |
| }, | |
| "file_extension": ".py", | |
| "mimetype": "text/x-python", | |
| "name": "python", | |
| "nbconvert_exporter": "python", | |
| "pygments_lexer": "ipython3", | |
| "version": "3.5.0" | |
| } | |
| }, | |
| "nbformat": 4, | |
| "nbformat_minor": 0 | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment