Repository navigation
Expand file tree
/
Copy pathaction_data.py
More file actions
146 lines (108 loc) · 4.31 KB
/
Copy pathaction_data.py
File metadata and controls
146 lines (108 loc) · 4.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
# we will merge the condition and control dataframes
# and we will create the action.csv file that will contain the features
# for every patient
# we will keep the features which correspond to the first timestamp with
# time 00:00:00 and we will remove the previous ones
# The dataframe will end at the last timestamp with time 00:00:00 - excluding this last 00:00:00
import pandas as pd
import numpy as np
data = pd.DataFrame()
# date has format: 2019-03-25 00:00:00
# we will keep only the time
# healthy = 1, depressed = 0
for i in range(1, 24):
file = 'Data/condition/condition_' + str(i) + '.csv'
df = pd.read_csv(file)
df['patient'] = i
df['target'] = 0
data = pd.concat([data, df], axis=0)
for i in range(1, 33):
file = 'Data/control/control_' + str(i) + '.csv'
df = pd.read_csv(file)
df['target'] = 1
df['patient'] = i+23
data = pd.concat([data, df], axis=0)
data = data.drop(['date'], axis=1)
#print(data)
data['timestamp'] = data['timestamp'].str[11:]
data['timestamp'] = pd.to_datetime(data['timestamp'], format='%H:%M:%S').dt.time
# we remove the rows before the first 00:00:00 and after the last 00:00:00 for each patient
data2 = pd.DataFrame()
for i in range(1, 56):
#print(data[data['patient'] == i])
df = data[data['patient'] == i]
patient = i
df = df.reset_index(drop=True)
for j in range(0, len(df)):
if df['timestamp'][j] == pd.to_datetime('00:00:00', format='%H:%M:%S').time():
first = j
break
for j in range(len(df)-1, -1, -1):
if df['timestamp'][j] == pd.to_datetime('23:59:00', format='%H:%M:%S').time():
last = j
break
df['patient'] = patient
df = df[first:last+1]
#df = df.reset_index(drop=True)
#data = data.drop(data[data['patient'] == i].index)
data2 = pd.concat([data2, df], axis=0)
data2 = data2.reset_index(drop=True)
data3 = pd.DataFrame()
for i in range(1, 56):
df = data2[data2['patient'] == i]
df = df.reset_index(drop=True)
df = df.groupby(np.arange(len(df))//30).mean(numeric_only=True)
# we keep rows multiple of 48
if len(df) % 48 != 0:
df = df[:-(len(df) % 48)]
df = df.round(3)
df['patient'] = i
#print(len(df))
data3 = pd.concat([data3, df], axis=0)
data3 = data3.reset_index(drop=True)
print(data3)
# one day has 1440 minutes and we have one measurement every 30 minutes
# so we have 48 measurements per day
# for every patient wi will keep the maximum amount of days that is divisible by 48
##################################################################
##################################################################
# find the index for the first and the last day for every patient
last_day = []
for i in range(1, 56):
df = data2[data2['patient'] == i]
df = df.reset_index(drop=True)
last_day.append(df.index[-1])
first_day = [0]
for i in range(1, 55):
first_day.append(first_day[i-1] + last_day[i-1] + 1)
print(first_day)
header = ['first_day']
# save the first day to a csv file
df = pd.DataFrame(first_day, columns=header)
df.to_csv('Data/first_row_for_each_patient.csv', index=False)
first_row = pd.read_csv('Data/first_row_for_each_patient.csv')
##################################################################
##################################################################
for i in range(0, len(data), 48):
data3.loc[i:i+48, 'patient_new'] = i/48 + 1
##################################################################
##################################################################
scores = pd.read_csv('Data/scores.csv')
#create a column with patient number for every patient in the scores dataframe
scores['patient'] = 0
for i in range(0, len(scores)):
scores.loc[i, 'patient'] = i+1
#print(scores)
# create a new column with afftype for every patient in the data3 dataframe and assign the corresponding afftype from the scores dataframe
data3['afftype'] = 0
for i in range(0, len(data3)):
patient = data3['patient'][i]
data3.loc[i, 'afftype'] = scores['afftype'][patient-1]
# if afftype is 3 then we assign 1
if data3['afftype'][i] == 3:
data3.loc[i, 'afftype'] = 1
# if afftype is NaN then we assign 0
if np.isnan(data3['afftype'][i]):
data3.loc[i, 'afftype'] = 0
# save the data to a csv file
data3.to_csv('Data/action.csv', index=False)