mirror of
https://github.com/wassname/Castor.git
synced 2026-10-03 12:00:15 +08:00
Ext feats bug fix (#19)
+ sm model no external features baseline + sm model with IDF weights + sm model with IDF weights without removing punctuation --> barely better than df/idf (a la Pytorch). + sm model with stemming before computing IDF weights ^ all on the TrecQA dataset
This commit is contained in:
1 parent
2758dee98f
commit
a67e2d12c4
9 files changed
+332
-72
No files matched your search
+6
-5
@@ -107,14 +107,15 @@ def read_in_dataset(dataset_folder, set_folder):
|
||||
len_s_list = [len(s.split()) for s in sentences]
|
||||
|
||||
labels = [int(line.strip()) for line in open(os.path.join(set_path, 'sim.txt')).readlines()]
|
||||
ext_feats = np.array([list(map(float, line.strip().split(' '))) \
|
||||
for line in open(os.path.join(set_path, 'overlap_feats.txt')).readlines()])
|
||||
|
||||
#y = torch.from_numpy(labels)
|
||||
#return questions, sentences, y
|
||||
# ext_feats = [np.zeros(4)] * len(questions)
|
||||
# if load_ext_features:
|
||||
# ext_feats = np.array([list(map(float, line.strip().split(' '))) \
|
||||
# for line in open(os.path.join(set_path, 'overlap_feats.txt')).readlines()])
|
||||
|
||||
vocab = [line.strip() for line in open(os.path.join(dataset_folder, 'vocab.txt')).readlines()]
|
||||
return questions, sentences, labels, vocab, max(len_q_list), max(len_s_list), ext_feats
|
||||
|
||||
return [questions, sentences, labels, max(len_q_list), max(len_s_list), vocab]
|
||||
|
||||
|
||||
def get_test_qids_labels(dataset_folder, set_folder):
|
||||
|
||||
Reference in new issue
Block a user