-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathload_data.py
More file actions
96 lines (82 loc) · 4.31 KB
/
Copy pathload_data.py
File metadata and controls
96 lines (82 loc) · 4.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
import numpy as np
import pandas as pd
import os
import random
import scipy
import scipy.io
import torch
from scipy.sparse import csc_matrix, coo_matrix, csr_matrix
from scipy.stats import ks_2samp, pearsonr, spearmanr #cramervonmises_2samp, entropy
from scipy.special import softmax as scp_softmax
from scipy.optimize import linear_sum_assignment
from sklearn.metrics import pairwise_distances
def load_data(base_dir, study, truncate_time_point = 12, gene_num_cap = 4000, cell_num_cap = 4000, gene_select = 'Threshold'):
if study.startswith('schiebinger2019'):
time_pt_list = np.arange(0, 8, 0.5)
num_time_pts = len(time_pt_list)
dense_mat_list = [] # This is a list, each element is a mat. Each mat could have totally different r and c.
for t in np.arange(len(time_pt_list)):
# print(t)
time_pt = time_pt_list[t]
str_time_pt = str(time_pt).replace('.', '_')
fname = os.path.join(base_dir, study, 'gene_exp_mat_time_' + \
str_time_pt + '_100k_sf_1e04_rc.mtx')
sparse_mat = scipy.io.mmread(fname) #coo format, only some couples of points. (rows, cols, value)
sparse_mat = coo_matrix.tocsr(sparse_mat) #easier to index
dense_mat = np.array(csr_matrix.todense(sparse_mat))
dense_mat = 2**dense_mat
# print(t, dense_mat)
dense_mat_list.append(dense_mat)
elif study.startswith('cao2019'):
time_pt_list = [9.5, 10.5, 11.5, 12.5, 13.5]
num_time_pts = len(time_pt_list)
dense_mat_list = []
for t in np.arange(len(time_pt_list)):
time_pt = time_pt_list[t]
str_time_pt = str(time_pt).replace('.', '_')
fname = os.path.join(base_dir, study, 'gene_exp_mat_time_' + \
str_time_pt + '_10k' + + '_sf_1e04_rc.mtx') #'_g1000.mtx')
sparse_mat = scipy.io.mmread(fname) #coo format
sparse_mat = coo_matrix.transpose(sparse_mat) #rows now correspond to samples
sparse_mat = coo_matrix.tocsr(sparse_mat) #easier to index
dense_mat = np.array(csr_matrix.todense(sparse_mat))
dense_mat_list.append(dense_mat)
print('Truncate time point (included in structure training):', truncate_time_point)
print('Last time point:', len(dense_mat_list) - 1)
print('Max number of time points (length):', len(dense_mat_list), '\n')
print(f'------First element of the dense mat: {len(dense_mat_list[0])}, {len(dense_mat_list[0][0])}\n')
num_cells_per_tp_list_pre = [np.size(dense_mat_list[i], 0) for i in np.arange(len(dense_mat_list))]
min_num_cells = min(num_cells_per_tp_list_pre)
print('Minimum number of cells over all time points:', min_num_cells, '\n')
num_tps_total = len(dense_mat_list)
print("Num total: " + str(num_tps_total))
# Before they have different size. Now unify them, and make them like (cell, time, gene) for further use.
# Threshold or dca?
print("Number sample")
new_dense_mat_list = []
old_dense_mat_list = dense_mat_list
for t in np.arange(len(time_pt_list)):
dense_mat = dense_mat_list[t]
cells_per_tp = np.shape(dense_mat)[0]
if cells_per_tp < cell_num_cap:
idcs = np.random.choice(cells_per_tp, size=min_num_cells, replace=True) # Add some points i.i.d.
else:
idcs = np.random.choice(cells_per_tp, size=min_num_cells, replace=False)
dense_mat = dense_mat[idcs, :]
new_dense_mat_list.append(dense_mat)
dense_mat_list = new_dense_mat_list
for t in range(truncate_time_point+1):
print(dense_mat_list[t].shape)
return dense_mat_list, dense_mat_list
# One hot embedding:-----------]
one_hot_mat_all_tps_list = []
for t in torch.arange(len(dense_mat_list)):
categorical_tp = t*torch.ones((cell_num_cap))
categorical_tp = categorical_tp.type(torch.long)
one_hot_tp = torch.nn.functional.one_hot(categorical_tp, num_classes= len(dense_mat_list))
one_hot_mat_all_tps_list.append(one_hot_tp.numpy())
one_hot_mat_all_tps = np.concatenate(one_hot_mat_all_tps_list, axis=0)
return dense_mat_list, one_hot_mat_all_tps
# cell * t * gene
# cell * t * latent_space
# Threshold??