Repository navigation
Expand file tree
/
Copy pathcreate_data.py
More file actions
106 lines (96 loc) · 3.03 KB
/
Copy pathcreate_data.py
File metadata and controls
106 lines (96 loc) · 3.03 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
import numpy as np
import sys
import csv
#import matplotlib.pyplot as plt
import math
import re
# when using VM instance for first time:
# import nltk
# nltk.download('stopwords')
from nltk.corpus import stopwords
import torch
import torch.nn as nn
import torch.nn.utils
import torch.nn.functional as F
from torch.nn.utils.rnn import pad_packed_sequence, pack_padded_sequence
import pickle
csv.field_size_limit(sys.maxsize)
input_file = 'data/all_data_v3_w_text.csv'
DOCUMENT_IND = 8
SECTIONS_START_IND = 9
OUTCOME_IND = 4
f = open("data/lsa_popular_words_all.txt", "r")
vocab = {}
vocab_ind2word = {}
index = 0
for line in f:
vocab[line.strip()] = index
vocab_ind2word[index] = line.strip()
index += 1
regex = re.compile('[^a-zA-Z]')
stopwords = stopwords.words('english')
'''
1 for mentioned, 0 for not mentioned, for each section
'''
def section_features(row):
features = []
for i in row[SECTIONS_START_IND:]:
if i == '':
features.append(0)
else:
features.append(float(i))
return np.array(features)
'''
Features: word counts, sections, and clustering info
'''
def feature_extractor(row):
features = np.zeros((len(vocab)))
# f1 = section_features(row)
for word in row[DOCUMENT_IND].split():
word = regex.sub('', word.lower())
if (word not in stopwords) and (word != '') and (word in vocab):
features[vocab[word]] += 1
return features
def generate_tensors(train_examples):
tensors_features = []
tensors_values = []
print("Extracting features...")
for row,value in train_examples:
features = feature_extractor(row)
tensors_features.append(features)
tensors_values.append(value)
tensors_features = torch.Tensor(tensors_features)
tensors_values = torch.Tensor(tensors_values)
torch.save(tensors_features, 'tensors_features.pt')
torch.save(tensors_values, 'tensors_values.pt')
#tensors_features = torch.load('tensors_features.pt')
#tensors_values = torch.load('tensors_values.pt')
return tensors_features, tensors_values
#######################################################################################
def create_examples():
examples = []
with open(input_file, encoding='ISO-8859-1') as input:
reader = csv.reader(input)
next(reader)
n = 0
for row in reader:
#if matrix[n] == 0:
value = 0
if "APPROVED" in row[OUTCOME_IND] or "CONCURRED" in row[OUTCOME_IND]:
value = 1
examples.append((row, value))
n += 1
examples = np.array(examples)
np.savetxt('examples.out', examples, '%s')
#examples = np.loadtxt('examples.out')
return examples
"""
def main(args):
examples = create_examples()
np.random.shuffle(examples)
# train_examples = examples[:7*num_rows//10]
# test_examples = examples[7*num_rows//10:]
train_examples = examples[:9*len(examples)//10]
test_examples = examples[9*len(examples)//10:]
generate_tensors(train_examples)
"""