Ext feats bug fix (#19)

+ sm model no external features baseline
+ sm model with IDF weights
+ sm model with IDF weights without removing punctuation --> barely better than df/idf (a la Pytorch).
+ sm model with stemming before computing IDF weights
^ all on the TrecQA dataset
This commit is contained in:
gauravbaruah authored and Jimmy Lin committed 2017-04-18 12:32:43 -04:00
1 parent 2758dee98f
commit a67e2d12c4
9 files changed
+332 -72

No files matched your search

+6 -5
View File
@@ -107,14 +107,15 @@ def read_in_dataset(dataset_folder, set_folder):
len_s_list = [len(s.split()) for s in sentences]
labels = [int(line.strip()) for line in open(os.path.join(set_path, 'sim.txt')).readlines()]
ext_feats = np.array([list(map(float, line.strip().split(' '))) \
for line in open(os.path.join(set_path, 'overlap_feats.txt')).readlines()])
#y = torch.from_numpy(labels)
#return questions, sentences, y
# ext_feats = [np.zeros(4)] * len(questions)
# if load_ext_features:
# ext_feats = np.array([list(map(float, line.strip().split(' '))) \
# for line in open(os.path.join(set_path, 'overlap_feats.txt')).readlines()])
vocab = [line.strip() for line in open(os.path.join(dataset_folder, 'vocab.txt')).readlines()]
return questions, sentences, labels, vocab, max(len_q_list), max(len_s_list), ext_feats
return [questions, sentences, labels, max(len_q_list), max(len_s_list), vocab]
def get_test_qids_labels(dataset_folder, set_folder):