dmlc--dgl
44089c8b4d
* Merge * [Graph][CUDA] Graph on GPU and many refactoring (#1791) * change edge_ids behavior and C++ impl * fix unittests; remove utils.Index in edge_id * pass mx and th tests * pass tf test * add aten::Scatter_ * Add nonzero; impl CSRGetDataAndIndices/CSRSliceMatrix * CSRGetData and CSRGetDataAndIndices passed tests * CSRSliceMatrix basic tests * fix bug in empty slice * CUDA CSRHasDuplicate * has_node; has_edge_between * predecessors, successors * deprecate send/recv; fix send_and_recv * deprecate send/recv; fix send_and_recv * in_edges; out_edges; all_edges; apply_edges * in deg/out deg * subgraph/edge_subgraph * adj * in_subgraph/out_subgraph * sample neighbors * set/get_n/e_repr * wip: working on refactoring all idtypes * pass ndata/edata tests on gpu * fix * stash * workaround nonzero issue * stash * nx conversion * test_hetero_basics except update routines * test_update_routines * test_hetero_basics for pytorch * more fixes * WIP: flatten graph * wip: flatten * test_flatten * test_to_device * fix bug in to_homo * fix bug in CSRSliceMatrix * pass subgraph test * fix send_and_recv * fix filter * test_heterograph * passed all pytorch tests * fix mx unittest * fix pytorch test_nn * fix all unittests for PyTorch * passed all mxnet tests * lint * fix tf nn test * pass all tf tests * lint * lint * change deprecation * try fix compile * lint * update METIDS * fix utest * fix * fix utests * try debug * revert * small fix * fix utests * upd * upd * upd * fix * upd * upd * upd * upd * upd * trigger * +1s * [kernel] Use heterograph index instead of unitgraph index (#1813) * upd * upd * upd * fix * upd * upd * upd * upd * upd * trigger * +1s * [Graph] Mutation for Heterograph (#1818) * mutation add_nodes and add_edges * Add support for remove_edges, remove_nodes, add_selfloop, remove_selfloop * Fix Co-authored-by: Ubuntu <ubuntu@ip-172-31-51-214.ec2.internal> * upd * upd * upd * fix * [Transfom] Mutable transform (#1833) * add nodesy * All three * Fix * lint * Add some test case * Fix * Fix * Fix * Fix * Fix * Fix * fix * triger * Fix * fix Co-authored-by: Ubuntu <ubuntu@ip-172-31-51-214.ec2.internal> * [Graph] Migrate Batch & Readout module to heterograph (#1836) * dgl.batch * unbatch * fix to device * reduce readout; segment reduce * change batch_num_nodes|edges to function * reduce readout/ softmax * broadcast * topk * fix * fix tf and mx * fix some ci * fix batch but unbatch differently * new checkk * upd * upd * upd * idtype behavior; code reorg * idtype behavior; code reorg * wip: test_basics * pass test_basics * WIP: from nx/ to nx * missing files * upd * pass test_basics:test_nx_conversion * Fix test * Fix inplace update * WIP: fixing tests * upd * pass test_transform cpu * pass gpu test_transform * pass test_batched_graph * GPU graph auto cast to int32 * missing file * stash * WIP: rgcn-hetero * Fix two datasety * upd * weird * Fix capsuley * fuck you * fuck matthias * Fix dgmg * fix bug in block degrees; pass rgcn-hetero * rgcn * gat and diffpool fix also fix ppi and tu dataset * Tree LSTM * pointcloud * rrn; wip: sgc * resolve conflicts * upd * sgc and reddit dataset * upd * Fix deepwalk, gindt and gcn * fix datasets and sign * optimization * optimization * upd * upd * Fix GIN * fix bug in add_nodes add_edges; tagcn * adaptive sampling and gcmc * upd * upd * fix geometric * fix * metapath2vec * fix agnn * fix pickling problem of block * fix utests * miss file * linegraph * upd * upd * upd * graphsage * stgcn_wave * fix hgt * on unittests * Fix transformer * Fix HAN * passed pytorch unittests * lint * fix * Fix cluster gcn * cluster-gcn is ready * on fixing block related codes * 2nd order derivative * Revert "2nd order derivative" This reverts commit 523bf6c249bee61b51b1ad1babf42aad4167f206. * passed torch utests again * fix all mxnet unittests * delete some useless tests * pass all tf cpu tests * disable * disable distributed unittest * fix * fix * lint * fix * fix * fix script * fix tutorial * fix apply edges bug * fix 2 basics * fix tutorial Co-authored-by: yzh119 <expye@outlook.com> Co-authored-by: xiang song(charlie.song) <classicxsong@gmail.com> Co-authored-by: Ubuntu <ubuntu@ip-172-31-51-214.ec2.internal> Co-authored-by: Ubuntu <ubuntu@ip-172-31-7-42.us-west-2.compute.internal> Co-authored-by: Ubuntu <ubuntu@ip-172-31-1-5.us-west-2.compute.internal> Co-authored-by: Ubuntu <ubuntu@ip-172-31-68-185.ec2.internal>
128 行
5.4 KiB
Python
128 行
5.4 KiB
Python
import os, json, tqdm
|
|
import numpy as np
|
|
import dgl
|
|
from zipfile import ZipFile
|
|
from torch.utils.data import Dataset
|
|
from scipy.sparse import csr_matrix
|
|
from dgl.data.utils import download, get_download_dir
|
|
|
|
class ShapeNet(object):
|
|
def __init__(self, num_points=2048, normal_channel=True):
|
|
self.num_points = num_points
|
|
self.normal_channel = normal_channel
|
|
|
|
SHAPENET_DOWNLOAD_URL = "https://shapenet.cs.stanford.edu/media/shapenetcore_partanno_segmentation_benchmark_v0_normal.zip"
|
|
download_path = get_download_dir()
|
|
data_filename = "shapenetcore_partanno_segmentation_benchmark_v0_normal.zip"
|
|
data_path = os.path.join(download_path, "shapenetcore_partanno_segmentation_benchmark_v0_normal")
|
|
if not os.path.exists(data_path):
|
|
local_path = os.path.join(download_path, data_filename)
|
|
if not os.path.exists(local_path):
|
|
download(SHAPENET_DOWNLOAD_URL, local_path, verify_ssl=False)
|
|
with ZipFile(local_path) as z:
|
|
z.extractall(path=download_path)
|
|
|
|
synset_file = "synsetoffset2category.txt"
|
|
with open(os.path.join(data_path, synset_file)) as f:
|
|
synset = [t.split('\n')[0].split('\t') for t in f.readlines()]
|
|
self.synset_dict = {}
|
|
for syn in synset:
|
|
self.synset_dict[syn[1]] = syn[0]
|
|
self.seg_classes = {'Airplane': [0, 1, 2, 3],
|
|
'Bag': [4, 5],
|
|
'Cap': [6, 7],
|
|
'Car': [8, 9, 10, 11],
|
|
'Chair': [12, 13, 14, 15],
|
|
'Earphone': [16, 17, 18],
|
|
'Guitar': [19, 20, 21],
|
|
'Knife': [22, 23],
|
|
'Lamp': [24, 25, 26, 27],
|
|
'Laptop': [28, 29],
|
|
'Motorbike': [30, 31, 32, 33, 34, 35],
|
|
'Mug': [36, 37],
|
|
'Pistol': [38, 39, 40],
|
|
'Rocket': [41, 42, 43],
|
|
'Skateboard': [44, 45, 46],
|
|
'Table': [47, 48, 49]}
|
|
|
|
train_split_json = 'shuffled_train_file_list.json'
|
|
val_split_json = 'shuffled_val_file_list.json'
|
|
test_split_json = 'shuffled_test_file_list.json'
|
|
split_path = os.path.join(data_path, 'train_test_split')
|
|
with open(os.path.join(split_path, train_split_json)) as f:
|
|
tmp = f.read()
|
|
self.train_file_list = [os.path.join(data_path, t.replace('shape_data/', '') + '.txt') for t in json.loads(tmp)]
|
|
with open(os.path.join(split_path, val_split_json)) as f:
|
|
tmp = f.read()
|
|
self.val_file_list = [os.path.join(data_path, t.replace('shape_data/', '') + '.txt') for t in json.loads(tmp)]
|
|
with open(os.path.join(split_path, test_split_json)) as f:
|
|
tmp = f.read()
|
|
self.test_file_list = [os.path.join(data_path, t.replace('shape_data/', '') + '.txt') for t in json.loads(tmp)]
|
|
|
|
def train(self):
|
|
return ShapeNetDataset(self, 'train', self.num_points, self.normal_channel)
|
|
|
|
def valid(self):
|
|
return ShapeNetDataset(self, 'valid', self.num_points, self.normal_channel)
|
|
|
|
def trainval(self):
|
|
return ShapeNetDataset(self, 'trainval', self.num_points, self.normal_channel)
|
|
|
|
def test(self):
|
|
return ShapeNetDataset(self, 'test', self.num_points, self.normal_channel)
|
|
|
|
class ShapeNetDataset(Dataset):
|
|
def __init__(self, shapenet, mode, num_points, normal_channel=True):
|
|
super(ShapeNetDataset, self).__init__()
|
|
self.mode = mode
|
|
self.num_points = num_points
|
|
if not normal_channel:
|
|
self.dim = 3
|
|
else:
|
|
self.dim = 6
|
|
|
|
if mode == 'train':
|
|
self.file_list = shapenet.train_file_list
|
|
elif mode == 'valid':
|
|
self.file_list = shapenet.val_file_list
|
|
elif mode == 'test':
|
|
self.file_list = shapenet.test_file_list
|
|
elif mode == 'trainval':
|
|
self.file_list = shapenet.train_file_list + shapenet.val_file_list
|
|
else:
|
|
raise "Not supported `mode`"
|
|
|
|
data_list = []
|
|
label_list = []
|
|
category_list = []
|
|
print('Loading data from split ' + self.mode)
|
|
for fn in tqdm.tqdm(self.file_list, ascii=True):
|
|
with open(fn) as f:
|
|
data = np.array([t.split('\n')[0].split(' ') for t in f.readlines()]).astype(np.float)
|
|
data_list.append(data[:, 0:self.dim])
|
|
label_list.append(data[:, 6].astype(np.int))
|
|
category_list.append(shapenet.synset_dict[fn.split('/')[-2]])
|
|
self.data = data_list
|
|
self.label = label_list
|
|
self.category = category_list
|
|
|
|
def translate(self, x, scale=(2/3, 3/2), shift=(-0.2, 0.2), size=3):
|
|
xyz1 = np.random.uniform(low=scale[0], high=scale[1], size=[size])
|
|
xyz2 = np.random.uniform(low=shift[0], high=shift[1], size=[size])
|
|
x = np.add(np.multiply(x, xyz1), xyz2).astype('float32')
|
|
return x
|
|
|
|
def __len__(self):
|
|
return len(self.data)
|
|
|
|
def __getitem__(self, i):
|
|
inds = np.random.choice(self.data[i].shape[0], self.num_points, replace=True)
|
|
x = self.data[i][inds,:self.dim]
|
|
y = self.label[i][inds]
|
|
cat = self.category[i]
|
|
if self.mode == 'train':
|
|
x = self.translate(x, size=self.dim)
|
|
x = x.astype(np.float)
|
|
y = y.astype(np.int)
|
|
return x, y, cat
|