{ "cells": [ { "cell_type": "code", "execution_count": 1, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "env: CUDA_VISIBLE_DEVICES=1\n" ] } ], "source": [ "%env CUDA_VISIBLE_DEVICES=1" ] }, { "cell_type": "code", "execution_count": 2, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "/home/pczapla/workspace/ulmfit-multilingual\n" ] } ], "source": [ "%reload_ext autoreload\n", "%autoreload 2\n", "%matplotlib inline\n", "%cd ..\n", "from fastai.text import *" ] }, { "cell_type": "code", "execution_count": 3, "metadata": {}, "outputs": [], "source": [ "from multifit.datasets import ULMFiTDataset, Dataset\n", "from multifit import ULMFiT, multifit1552_fp16, multifit1552_fp32" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# Wikipedia Pretraining" ] }, { "cell_type": "code", "execution_count": 4, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "def multifit1552_fp16():\n", " return multifit1552_fp32(bs=128).replace_(fp16=True, name=multifit1552_fp16.__name__)\n", "\n", "def multifit1552_fp32(bs=64):\n", " self = ULMFiT()\n", " self.replace_(\n", " label_smoothing_eps=0.0,\n", " true_wd=True,\n", " wd=0.1,\n", " seed=0,\n", " fp16=False,\n", " bs=bs,\n", " use_adam_08=False,\n", " early_stopping=None,\n", " name=multifit1552_fp32.__name__\n", " )\n", " self.arch.replace_(\n", " tokenizer='fsp',\n", " max_vocab=15000,\n", " qrnn=True,\n", " n_layers=4,\n", " n_hid=1552\n", " )\n", " self.pretrain_lm.replace_(num_epochs=10, drop_mult=0.5, lr=(1e-2 * bs / 48))\n", " self.finetuine_lm.replace_(num_epochs=10, drop_mult=1.0, lr=(1e-3 * bs / 48))\n", " self.classifier.replace_(num_epochs=8, drop_mult=0.5, bs=20, label_smoothing_eps=0.1)\n", " return self\n", "\n" ] } ], "source": [ "import inspect\n", "print(inspect.getsource(multifit1552_fp16))\n", "print(inspect.getsource(multifit1552_fp32))" ] }, { "cell_type": "code", "execution_count": 5, "metadata": {}, "outputs": [], "source": [ "exp = multifit1552_fp16()" ] }, { "cell_type": "code", "execution_count": 6, "metadata": {}, "outputs": [], "source": [ "wiki_dataset = exp.arch.dataset(Path(f'data/wiki/ja-100'))" ] }, { "cell_type": "code", "execution_count": 7, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Data lm-notst, trn: 120037, val: 63\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n" ] }, { "data": { "text/html": [ "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
idxtext
0▁月 ▁9 ▁日 ▁) ▁は ▁、 ▁日 ▁本 ▁の ▁ 漫 ▁画 ▁家 ▁、 ▁ アニメーター ▁、 ▁ アニメーション ▁監 ▁ 督 ▁。 ▁大 ▁ 阪 ▁ 帝 ▁国 ▁大 ▁学 ▁ 附 ▁ 属 ▁ 医 ▁学 ▁ 専 ▁ 門 ▁部 ▁を ▁ 卒 ▁業 ▁、 ▁ 医 ▁ 師 ▁ 免 ▁ 許 ▁取 ▁ 得 ▁、 ▁ のち ▁ 医 ▁学 ▁ 博 ▁ 士 ▁(
1▁では ▁小 ▁学 ▁校 ▁の ▁教 ▁ 室 ▁が ▁ 不 ▁ 足 ▁するなど ▁、 ▁人 ▁ 口 ▁が ▁ 急 ▁ 増 ▁している ▁。 ▁これらの ことから ▁、 ▁ 老 ▁年 ▁人 ▁ 口 ▁の ▁ 比 ▁ 率 ▁が ▁大 ▁ 阪 ▁市 ▁24 ▁区 ▁の ▁中 ▁で ▁最 ▁も ▁ 低 ▁くなっている ▁。 ▁地 ▁理 ▁. ▁区 ▁ 域 ▁は ▁東 ▁を ▁西 ▁ 横 ▁ 堀 ▁川
2トラック などに ▁手 ▁ 荷 ▁物 ▁が ▁ 残 ▁ っ ていても それが ▁ 忘 ▁れ ▁物 ▁である とは ▁ 認 ▁ 識 ▁しない ▁ 可 ▁ 能 ▁性 ▁が ▁高 ▁い ▁。 ▁また ▁ 忘 ▁れ ▁物 ▁として ▁手 ▁ 荷 ▁物 ▁を ▁発 ▁見 ▁し ていても ▁、 ▁それが ▁ 爆 ▁発 ▁物 ▁である ▁事 ▁には ▁ 気 ▁ 付 ▁かない ▁ 可 ▁ 能 ▁性 ▁もあるが ▁、 ▁
3郎 ▁は ▁ 淀 ▁野 ▁と ▁ 近 ▁所 ▁を ▁ 散 ▁ 歩 ▁中 ▁、 ▁「 ▁東 ▁京 ▁の ▁ 横 ▁ 光 ▁は どう や ▁? ▁」 ▁と ▁ 質 ▁ 問 ▁し ▁、 ▁ 勢 ▁いの あった ▁ 横 ▁ 光 ▁ 利 ▁一 ▁を ライバル ▁ 視 ▁していた ▁。 ▁12 ▁月 ▁下 ▁ 旬 ▁、 ▁ 母 ▁が ▁ 阿 ▁ 倍 ▁野 ▁の ▁小 ▁間
4▁ 速 ▁に アラブ ▁化 ▁・ ▁ イスラム ▁化 ▁した ▁。 ▁8 ▁世 ▁ 紀 ▁には アッ バース ▁ 朝 ▁の カリフ が バグ ダー ド に ▁ 都 ▁を ▁ 造 ▁ 営 ▁し ▁、 ▁ アッ バース ▁ 朝 ▁が ▁ 滅 ▁びる まで イスラム ▁世 ▁ 界 ▁の ▁ 精 ▁神 ▁的 ▁中 ▁ 心 ▁として ▁ 栄 ▁えた ▁。 ▁10 ▁世 ▁ 紀 ▁ 末 ▁に
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "wiki_dataset.load_lm_databunch(bs=128,bptt=70).show_batch()" ] }, { "cell_type": "code", "execution_count": 10, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting LM weights seed seed to 0\n", "Training args: {'drop_mult': 0.5, 'true_wd': True, 'wd': 0.1, 'pretrained': False} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.1, 'hidden_p': 0.15, 'input_p': 0.25, 'embed_p': 0.02, 'weight_p': 0.2, 'tie_weights': True, 'out_bias': True}\n", "Data lm-notst, trn: 120037, val: 63\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Setting LM training seed seed to 0\n", "Experiment data/wiki/ja-100/models/fsp15k/multifit1552_fp16\n", "Training lm from random weights\n" ] }, { "data": { "text/html": [ "Total time: 9:30:58

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
03.1076743.0973750.41180257:20
13.3310743.3642280.38291757:24
24.5642334.4705730.28338157:10
33.6284113.5523230.36470856:55
43.4899293.4597490.37072156:50
53.3106853.3335060.38370256:54
63.1548213.1799650.40172657:02
72.9726262.9314680.43252357:05
82.8314862.7140950.46195957:06
92.7606422.5984360.48106257:07
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "this Learner object self-destroyed - it still exists, but no longer usable\n", "Language model saved to data/wiki/ja-100/models/fsp15k/multifit1552_fp16\n", "Saving dump to data/wiki/ja-100/models/fsp15k/multifit1552_fp16/pretraining.json\n" ] } ], "source": [ "exp.pretrain_lm.train_(wiki_dataset)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# MLDoc Trainig" ] }, { "cell_type": "code", "execution_count": 7, "metadata": {}, "outputs": [], "source": [ "mldoc_dataset = exp.arch.dataset(Path('data/mldoc/ja-1'))" ] }, { "cell_type": "code", "execution_count": 8, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n" ] }, { "data": { "text/html": [ "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
texttarget
▁xxbos ▁xxfld ▁1 ▁ [ ▁東 ▁京 ▁25 ▁日 ▁ ロイ ター ▁ ] ▁野 ▁ 村 ▁ 総 ▁ 研 ▁が ▁、 ▁9 ▁月 ▁24 ▁日 ▁ 付 ▁で まとめた ▁ 香 ▁ 港 ▁・ ▁ シン ガ ポー ▁ ル ▁・ ▁ マレーシア ▁市 ▁場 ▁の ▁主 ▁要 ▁株 ▁式 ▁の ▁ 投 ▁ 資 ▁ 判 ▁ 断 ▁は ▁、 ▁以 ▁下 ▁の ▁通 ▁り ▁。 ▁(0
▁xxbos ▁xxfld ▁1 ▁ [ ▁東 ▁京 ▁25 ▁日 ▁ ロイ ター ▁ ] ▁野 ▁ 村 ▁ 総 ▁ 研 ▁が ▁、 ▁11 ▁月 ▁25 ▁日 ▁ 付 ▁で まとめた ▁ 香 ▁ 港 ▁・ ▁ シン ガ ポ ▁ ール ▁・ ▁ マレーシア ▁市 ▁場 ▁の ▁主 ▁要 ▁株 ▁式 ▁の ▁ 投 ▁ 資 ▁ 判 ▁ 断 ▁は ▁、 ▁以 ▁下 ▁の ▁通 ▁り ▁。 ▁(0
▁xxbos ▁xxfld ▁1 ▁ 直 ▁ 近 ▁上 ▁ 値 ▁ 抵 ▁ 抗 ▁ 短 ▁期 ▁上 ▁ 値 ▁ 抵 ▁ 抗 ▁ 直 ▁ 近 ▁下 ▁ 値 ▁ 支 ▁ 持 ▁ 短 ▁期 ▁下 ▁ 値 ▁ 支 ▁ 持 ▁ ドル ▁ / ▁ 円 ▁ 125 ▁. ▁ 60 ▁12 6 ▁. ▁00 ▁ 124 ▁. ▁ 60 ▁ 124 ▁. ▁ 403
▁xxbos ▁xxfld ▁1 ▁ [ ▁東 ▁京 ▁26 ▁日 ▁ ロイ ター ▁ ] ▁ 今 ▁ 週 ▁は ▁、 ▁28 ▁日 ▁発 ▁表 ▁の 8 ▁月 ▁ 調 ▁ 査 ▁日 ▁ 銀 ▁ 短 ▁ 観 ▁( ▁ 企 ▁業 ▁ 短 ▁期 ▁経 ▁ 済 ▁ 観 ▁ 測 ▁ 調 ▁ 査 ▁) ▁、 ▁30 ▁日 ▁に ▁発 ▁表 ▁される 7 ▁月 ▁ 鉱 ▁ 工3
▁xxbos ▁xxfld ▁1 ▁ レート は ▁ 終 ▁ 値 ▁( ▁前 ▁日 ▁ 比 ▁または ▁前 ▁ 週 ▁ 末 ▁ 比 ▁) ▁、 ▁ 安 ▁ 値 ▁ 〜 ▁高 ▁ 値 ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━ ▁ ━3
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" } ], "source": [ "mldoc_dataset.load_clas_databunch(bs=128)[0].show_batch()" ] }, { "cell_type": "code", "execution_count": 12, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting LM weights seed seed to 0\n", "Training args: {'drop_mult': 1.0, 'true_wd': True, 'wd': 0.1, 'pretrained': True, 'pretrained_fnames': [PosixPath('/home/pczapla/workspace/ulmfit-multilingual/data/wiki/ja-100/models/fsp15k/multifit1552_fp16/lm_best'), PosixPath('/home/pczapla/workspace/ulmfit-multilingual/data/wiki/ja-100/models/fsp15k/itos')]} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.1, 'hidden_p': 0.15, 'input_p': 0.25, 'embed_p': 0.02, 'weight_p': 0.2, 'tie_weights': True, 'out_bias': True}\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Setting LM training seed seed to 0\n", "Experiment data/mldoc/ja-1/models/fsp15k/multifit1552_fp16\n", "Fitting using 2 cycle fit schedule\n" ] }, { "data": { "text/html": [ "Total time: 01:26

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
02.2198601.8335440.60693301:26
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "data": { "text/html": [ "Total time: 18:17

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
01.9445961.6707730.63707201:49
11.8063801.5286840.66275301:49
21.7205331.4449820.67730501:49
31.6633991.3798560.68837601:49
41.6180901.3347220.69601901:49
51.5722661.2920200.70429101:49
61.5509541.2545810.71130201:49
71.4744051.2268870.71628701:49
81.4462001.2123170.71896801:49
91.4454661.2095160.71954601:50
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "this Learner object self-destroyed - it still exists, but no longer usable\n", "Language model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/finetuining.json\n" ] } ], "source": [ "exp.finetune_lm.train_(mldoc_dataset)" ] }, { "cell_type": "code", "execution_count": 9, "metadata": { "scrolled": true }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Loading data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/finetuining.json\n", "Loading data/wiki/ja-100/models/fsp15k/multifit1552_fp16/pretraining.json\n" ] }, { "data": { "text/plain": [ "ULMFiTFinetuining(seed=0, name='multifit1552_fp16', experiment_path=PosixPath('data/mldoc/ja-1/models/fsp15k/multifit1552_fp16'), dataset_path=PosixPath('data/mldoc/ja-1'), num_epochs=10, bs=128, bptt=70, drop_mult=1.0, label_smoothing_eps=0.0, use_adam_08=False, true_wd=True, wd=0.1, fp16=True, lr=0.0026666666666666666, pretrained=True)" ] }, "execution_count": 9, "metadata": {}, "output_type": "execute_result" } ], "source": [ "exp.load_(Path('data/mldoc/ja-1/models/fsp15k/multifit1552_fp16')).finetune_lm" ] }, { "cell_type": "code", "execution_count": 10, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 0\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 0\n" ] }, { "data": { "text/html": [ "Total time: 01:52

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.9119670.6398160.85000000:14
10.7813250.6461440.87400000:13
20.7034130.7223810.84900000:13
30.6220300.6786570.87500000:13
40.5456900.6236890.89000000:13
50.4914530.6282240.88300000:14
60.4518030.6285840.88600000:14
70.4290280.6272610.88800000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=0)" ] }, { "cell_type": "code", "execution_count": 11, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 1\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 1\n" ] }, { "data": { "text/html": [ "Total time: 01:54

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.9084080.6543190.86000000:14
10.7903980.7684380.81200000:14
20.6928330.6503520.88300000:14
30.6079500.6597730.86700000:14
40.5490190.6271780.88600000:14
50.4977830.7014990.87900000:14
60.4558030.6519050.87000000:14
70.4300480.6347290.87900000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed1\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed1/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=1)" ] }, { "cell_type": "code", "execution_count": 12, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 2\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 2\n" ] }, { "data": { "text/html": [ "Total time: 01:54

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.9251130.6533450.84900000:14
10.7748870.6329420.87600000:14
20.6862290.6614850.87000000:14
30.6303400.6306090.87700000:14
40.5507890.6181770.87800000:14
50.4943870.6368920.87800000:14
60.4505000.6330690.88500000:14
70.4310310.6370730.88100000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed2\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed2/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=2)" ] }, { "cell_type": "code", "execution_count": 13, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 3\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 3\n" ] }, { "data": { "text/html": [ "Total time: 01:54

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.9036410.6161800.87200000:14
10.7980580.6117680.88200000:14
20.6921930.6740290.84200000:14
30.6221520.6058410.88900000:14
40.5459500.6405840.87700000:14
50.4957170.6252050.88500000:14
60.4541270.6293660.88500000:14
70.4309540.6272620.89100000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed3\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed3/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=3)" ] }, { "cell_type": "code", "execution_count": 14, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 4\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 4\n" ] }, { "data": { "text/html": [ "Total time: 01:53

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.8850450.6230220.86300000:14
10.7914880.6756740.85000000:14
20.7022590.7964260.83400000:14
30.6262440.6815710.86500000:14
40.5589770.6331190.89000000:14
50.4939250.6461560.87500000:14
60.4503240.6309830.88500000:14
70.4283130.6499530.87900000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed4\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed4/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=4)" ] }, { "cell_type": "code", "execution_count": 15, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 5\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 5\n" ] }, { "data": { "text/html": [ "Total time: 01:53

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.8986960.6347230.86500000:14
10.7878590.6289630.87800000:14
20.7009930.6424090.85500000:14
30.6197370.6320980.88000000:14
40.5587490.7293760.86100000:14
50.4895550.6298670.87700000:14
60.4517240.6444910.87600000:14
70.4295370.6511780.87900000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed5\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed5/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=5)" ] }, { "cell_type": "code", "execution_count": 16, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 6\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 6\n" ] }, { "data": { "text/html": [ "Total time: 01:53

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.8912800.6565610.86100000:14
10.7710490.6488310.85900000:14
20.6777180.6814190.85900000:14
30.6224750.6354180.87200000:14
40.5552760.6394900.87500000:14
50.4928250.6532720.87300000:14
60.4493470.6496540.87300000:14
70.4258220.6392830.87400000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed6\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed6/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=6)" ] }, { "cell_type": "code", "execution_count": 17, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Setting Classifier weights seed seed to 7\n", "Data lm-notst, trn: 9900, val: 1100\n", "Size of vocabulary: 15000\n", "First 20 words in vocab: ['▁xxunk', '▁xxpad', '▁xxbos', '▁xxeos', '▁xxfld', '▁xxmaj', '▁xxup', '▁xxrep', '▁xxwrep', '', '▁', '▁、', '▁。', '▁の', '▁に', '▁を', '▁年', '▁は', '▁・', '▁(']\n", "Data cls, trn: 1000, val: 1000\n", "Data tst, trn: 1000, val: 4000\n", "Training args: {'drop_mult': 0.5, 'wd': 0.1, 'pretrained': False, 'bptt': 70, 'loss_func': FlattenedLoss of LabelSmoothingCrossEntropy()} config: {'emb_sz': 400, 'n_hid': 1552, 'n_layers': 4, 'pad_token': 1, 'qrnn': True, 'bidir': False, 'output_p': 0.4, 'hidden_p': 0.3, 'input_p': 0.4, 'embed_p': 0.05, 'weight_p': 0.5}\n", "Loading pretrained model /home/pczapla/workspace/ulmfit-multilingual/data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/enc_best\n", "Setting Classifier training seed seed to 7\n" ] }, { "data": { "text/html": [ "Total time: 01:53

\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
epochtrain_lossvalid_lossaccuracytime
00.8902060.6181650.86900000:14
10.7871460.6148320.88400000:14
20.6957660.7056360.84300000:14
30.6251740.6118880.89000000:14
40.5611640.6779920.85600000:14
50.4961240.6235920.88100000:14
60.4541880.6477160.87600000:14
70.4318550.6334120.87900000:14
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "name": "stdout", "output_type": "stream", "text": [ "Classifier model saved to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed7\n", "Saving dump to data/mldoc/ja-1/models/fsp15k/multifit1552_fp16seed7/classifier.json\n", "this Learner object self-destroyed - it still exists, but no longer usable\n" ] } ], "source": [ "exp.classifier.train_(seed=7)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ "# Results" ] }, { "cell_type": "code", "execution_count": 9, "metadata": { "scrolled": false }, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "{'test loss': 0.5893779397010803, 'test FBeta': 0.9026554822921753, 'test Precision': 0.9043030738830566, 'test Recall': 0.902697741985321, 'test accuracy': 0.903249979019165, 'name': 'multifit1552_fp16seed1', 'valid loss': 0.6347294449806213, 'valid FBeta': 0.8822355270385742, 'valid Precision': 0.8312159776687622, 'valid Recall': 0.8805031180381775, 'valid accuracy': 0.9299362897872925, 'train loss': 0.3954845070838928, 'train FBeta': 0.9960629940032959, 'train Precision': 0.9978118538856506, 'train Recall': 1.0, 'train accuracy': 0.9980732202529907}\n", "Loading data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/classifier.json\n", "Loading data/mldoc/ja-1/models/fsp15k/multifit1552_fp16/finetuining.json\n", "Loading data/wiki/ja-100/models/fsp15k/multifit1552_fp16/pretraining.json\n" ] }, { "data": { "text/html": [ "

\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
nameseedtest accuracyvalid accuracy
5multifit1552_fp16seed440.896750.918580
1multifit1552_fp16seed660.898000.921776
3multifit1552_fp16seed550.897750.923404
2multifit1552_fp16seed770.902000.927660
0multifit1552_fp16seed220.902750.929336
6multifit1552_fp16seed110.903250.929936
7multifit1552_fp1600.899500.936170
4multifit1552_fp16seed330.902500.937634
\n", "
" ], "text/plain": [ " name seed test accuracy valid accuracy\n", "5 multifit1552_fp16seed4 4 0.89675 0.918580\n", "1 multifit1552_fp16seed6 6 0.89800 0.921776\n", "3 multifit1552_fp16seed5 5 0.89775 0.923404\n", "2 multifit1552_fp16seed7 7 0.90200 0.927660\n", "0 multifit1552_fp16seed2 2 0.90275 0.929336\n", "6 multifit1552_fp16seed1 1 0.90325 0.929936\n", "7 multifit1552_fp16 0 0.89950 0.936170\n", "4 multifit1552_fp16seed3 3 0.90250 0.937634" ] }, "execution_count": 9, "metadata": {}, "output_type": "execute_result" } ], "source": [ "def get_results(exp_path):\n", " exp = ULMFiT().load_(exp_path).classifier\n", " results = exp.validate() \n", " results.update(seed=exp.seed, fp16=exp.fp16)\n", " return results\n", "results = [get_results(exp_path) for exp_path in Path('data/mldoc/ja-1/models/fsp15k').glob(\"multifit1552_fp16*\")]\n", "results_df = pd.DataFrame.from_records(results)\n", "results_df.sort_values([\"valid accuracy\"])[[\"name\", \"seed\", \"test accuracy\", \"valid accuracy\"]]" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.7.2" } }, "nbformat": 4, "nbformat_minor": 2 }