chore: import upstream snapshot with attribution
This commit is contained in:
@@ -0,0 +1,34 @@
|
||||
# Relational-GCN
|
||||
|
||||
* Paper: [https://arxiv.org/abs/1703.06103](https://arxiv.org/abs/1703.06103)
|
||||
* Author's code for entity classification: [https://github.com/tkipf/relational-gcn](https://github.com/tkipf/relational-gcn)
|
||||
* Author's code for link prediction: [https://github.com/MichSchli/RelationPrediction](https://github.com/MichSchli/RelationPrediction)
|
||||
|
||||
### Dependencies
|
||||
* Tensorflow 2.2+
|
||||
* requests
|
||||
* rdflib
|
||||
* pandas
|
||||
|
||||
```
|
||||
pip install requests tensorflow rdflib pandas
|
||||
export DGLBACKEND=tensorflow
|
||||
```
|
||||
|
||||
Example code was tested with rdflib 4.2.2 and pandas 0.23.4
|
||||
|
||||
### Entity Classification
|
||||
AIFB: accuracy 92.78% (5 runs, DGL), 95.83% (paper)
|
||||
```
|
||||
python3 entity_classify.py -d aifb --testing --gpu 0
|
||||
```
|
||||
|
||||
MUTAG: accuracy 71.47% (5 runs, DGL), 73.23% (paper)
|
||||
```
|
||||
python3 entity_classify.py -d mutag --l2norm 5e-4 --n-bases 30 --testing --gpu 0
|
||||
```
|
||||
|
||||
BGS: accuracy 93.10% (5 runs, DGL n-base=25), 83.10% (paper n-base=40)
|
||||
```
|
||||
python3 entity_classify.py -d bgs --l2norm 5e-4 --n-bases 25 --testing --gpu 0
|
||||
```
|
||||
@@ -0,0 +1,280 @@
|
||||
"""
|
||||
Modeling Relational Data with Graph Convolutional Networks
|
||||
Paper: https://arxiv.org/abs/1703.06103
|
||||
Code: https://github.com/tkipf/relational-gcn
|
||||
|
||||
Difference compared to tkipf/relation-gcn
|
||||
* l2norm applied to all weights
|
||||
* remove nodes that won't be touched
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import time
|
||||
from functools import partial
|
||||
|
||||
import dgl
|
||||
|
||||
import numpy as np
|
||||
import tensorflow as tf
|
||||
from dgl.data.rdf import AIFBDataset, AMDataset, BGSDataset, MUTAGDataset
|
||||
from dgl.nn.tensorflow import RelGraphConv
|
||||
from model import BaseRGCN
|
||||
from tensorflow.keras import layers
|
||||
|
||||
|
||||
class EntityClassify(BaseRGCN):
|
||||
def create_features(self):
|
||||
features = tf.range(self.num_nodes)
|
||||
return features
|
||||
|
||||
def build_input_layer(self):
|
||||
return RelGraphConv(
|
||||
self.num_nodes,
|
||||
self.h_dim,
|
||||
self.num_rels,
|
||||
"basis",
|
||||
self.num_bases,
|
||||
activation=tf.nn.relu,
|
||||
self_loop=self.use_self_loop,
|
||||
dropout=self.dropout,
|
||||
)
|
||||
|
||||
def build_hidden_layer(self, idx):
|
||||
return RelGraphConv(
|
||||
self.h_dim,
|
||||
self.h_dim,
|
||||
self.num_rels,
|
||||
"basis",
|
||||
self.num_bases,
|
||||
activation=tf.nn.relu,
|
||||
self_loop=self.use_self_loop,
|
||||
dropout=self.dropout,
|
||||
)
|
||||
|
||||
def build_output_layer(self):
|
||||
return RelGraphConv(
|
||||
self.h_dim,
|
||||
self.out_dim,
|
||||
self.num_rels,
|
||||
"basis",
|
||||
self.num_bases,
|
||||
activation=partial(tf.nn.softmax, axis=1),
|
||||
self_loop=self.use_self_loop,
|
||||
)
|
||||
|
||||
|
||||
def acc(logits, labels, mask):
|
||||
logits = tf.gather(logits, mask)
|
||||
labels = tf.gather(labels, mask)
|
||||
indices = tf.math.argmax(logits, axis=1)
|
||||
acc = tf.reduce_mean(tf.cast(indices == labels, dtype=tf.float32))
|
||||
return acc
|
||||
|
||||
|
||||
def main(args):
|
||||
# load graph data
|
||||
if args.dataset == "aifb":
|
||||
dataset = AIFBDataset()
|
||||
elif args.dataset == "mutag":
|
||||
dataset = MUTAGDataset()
|
||||
elif args.dataset == "bgs":
|
||||
dataset = BGSDataset()
|
||||
elif args.dataset == "am":
|
||||
dataset = AMDataset()
|
||||
else:
|
||||
raise ValueError()
|
||||
|
||||
# preprocessing in cpu
|
||||
with tf.device("/cpu:0"):
|
||||
# Load from hetero-graph
|
||||
hg = dataset[0]
|
||||
|
||||
num_rels = len(hg.canonical_etypes)
|
||||
category = dataset.predict_category
|
||||
num_classes = dataset.num_classes
|
||||
train_mask = hg.nodes[category].data.pop("train_mask")
|
||||
test_mask = hg.nodes[category].data.pop("test_mask")
|
||||
train_idx = tf.squeeze(tf.where(train_mask))
|
||||
test_idx = tf.squeeze(tf.where(test_mask))
|
||||
labels = hg.nodes[category].data.pop("labels")
|
||||
|
||||
# split dataset into train, validate, test
|
||||
if args.validation:
|
||||
val_idx = train_idx[: len(train_idx) // 5]
|
||||
train_idx = train_idx[len(train_idx) // 5 :]
|
||||
else:
|
||||
val_idx = train_idx
|
||||
|
||||
# calculate norm for each edge type and store in edge
|
||||
for canonical_etype in hg.canonical_etypes:
|
||||
u, v, eid = hg.all_edges(form="all", etype=canonical_etype)
|
||||
_, inverse_index, count = tf.unique_with_counts(v)
|
||||
degrees = tf.gather(count, inverse_index)
|
||||
norm = tf.ones(eid.shape[0]) / tf.cast(degrees, tf.float32)
|
||||
norm = tf.expand_dims(norm, 1)
|
||||
hg.edges[canonical_etype].data["norm"] = norm
|
||||
|
||||
# get target category id
|
||||
category_id = len(hg.ntypes)
|
||||
for i, ntype in enumerate(hg.ntypes):
|
||||
if ntype == category:
|
||||
category_id = i
|
||||
|
||||
# edge type and normalization factor
|
||||
g = dgl.to_homogeneous(hg, edata=["norm"])
|
||||
|
||||
# check cuda
|
||||
if args.gpu < 0:
|
||||
device = "/cpu:0"
|
||||
use_cuda = False
|
||||
else:
|
||||
device = "/gpu:{}".format(args.gpu)
|
||||
g = g.to(device)
|
||||
use_cuda = True
|
||||
num_nodes = g.number_of_nodes()
|
||||
node_ids = tf.range(num_nodes, dtype=tf.int64)
|
||||
edge_norm = g.edata["norm"]
|
||||
edge_type = tf.cast(g.edata[dgl.ETYPE], tf.int64)
|
||||
|
||||
# find out the target node ids in g
|
||||
node_tids = g.ndata[dgl.NTYPE]
|
||||
loc = node_tids == category_id
|
||||
target_idx = tf.squeeze(tf.where(loc))
|
||||
|
||||
# since the nodes are featureless, the input feature is then the node id.
|
||||
feats = tf.range(num_nodes, dtype=tf.int64)
|
||||
|
||||
with tf.device(device):
|
||||
# create model
|
||||
model = EntityClassify(
|
||||
num_nodes,
|
||||
args.n_hidden,
|
||||
num_classes,
|
||||
num_rels,
|
||||
num_bases=args.n_bases,
|
||||
num_hidden_layers=args.n_layers - 2,
|
||||
dropout=args.dropout,
|
||||
use_self_loop=args.use_self_loop,
|
||||
use_cuda=use_cuda,
|
||||
)
|
||||
|
||||
# optimizer
|
||||
optimizer = tf.keras.optimizers.Adam(learning_rate=args.lr)
|
||||
# training loop
|
||||
print("start training...")
|
||||
forward_time = []
|
||||
backward_time = []
|
||||
loss_fcn = tf.keras.losses.SparseCategoricalCrossentropy(
|
||||
from_logits=False
|
||||
)
|
||||
for epoch in range(args.n_epochs):
|
||||
t0 = time.time()
|
||||
with tf.GradientTape() as tape:
|
||||
logits = model(g, feats, edge_type, edge_norm)
|
||||
logits = tf.gather(logits, target_idx)
|
||||
loss = loss_fcn(
|
||||
tf.gather(labels, train_idx), tf.gather(logits, train_idx)
|
||||
)
|
||||
# Manually Weight Decay
|
||||
# We found Tensorflow has a different implementation on weight decay
|
||||
# of Adam(W) optimizer with PyTorch. And this results in worse results.
|
||||
# Manually adding weights to the loss to do weight decay solves this problem.
|
||||
for weight in model.trainable_weights:
|
||||
loss = loss + args.l2norm * tf.nn.l2_loss(weight)
|
||||
t1 = time.time()
|
||||
grads = tape.gradient(loss, model.trainable_weights)
|
||||
optimizer.apply_gradients(zip(grads, model.trainable_weights))
|
||||
t2 = time.time()
|
||||
|
||||
forward_time.append(t1 - t0)
|
||||
backward_time.append(t2 - t1)
|
||||
print(
|
||||
"Epoch {:05d} | Train Forward Time(s) {:.4f} | Backward Time(s) {:.4f}".format(
|
||||
epoch, forward_time[-1], backward_time[-1]
|
||||
)
|
||||
)
|
||||
train_acc = acc(logits, labels, train_idx)
|
||||
val_loss = loss_fcn(
|
||||
tf.gather(labels, val_idx), tf.gather(logits, val_idx)
|
||||
)
|
||||
val_acc = acc(logits, labels, val_idx)
|
||||
print(
|
||||
"Train Accuracy: {:.4f} | Train Loss: {:.4f} | Validation Accuracy: {:.4f} | Validation loss: {:.4f}".format(
|
||||
train_acc,
|
||||
loss.numpy().item(),
|
||||
val_acc,
|
||||
val_loss.numpy().item(),
|
||||
)
|
||||
)
|
||||
print()
|
||||
|
||||
logits = model(g, feats, edge_type, edge_norm)
|
||||
logits = tf.gather(logits, target_idx)
|
||||
test_loss = loss_fcn(
|
||||
tf.gather(labels, test_idx), tf.gather(logits, test_idx)
|
||||
)
|
||||
test_acc = acc(logits, labels, test_idx)
|
||||
print(
|
||||
"Test Accuracy: {:.4f} | Test loss: {:.4f}".format(
|
||||
test_acc, test_loss.numpy().item()
|
||||
)
|
||||
)
|
||||
print()
|
||||
|
||||
print(
|
||||
"Mean forward time: {:4f}".format(
|
||||
np.mean(forward_time[len(forward_time) // 4 :])
|
||||
)
|
||||
)
|
||||
print(
|
||||
"Mean backward time: {:4f}".format(
|
||||
np.mean(backward_time[len(backward_time) // 4 :])
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="RGCN")
|
||||
parser.add_argument(
|
||||
"--dropout", type=float, default=0, help="dropout probability"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--n-hidden", type=int, default=16, help="number of hidden units"
|
||||
)
|
||||
parser.add_argument("--gpu", type=int, default=-1, help="gpu")
|
||||
parser.add_argument("--lr", type=float, default=1e-2, help="learning rate")
|
||||
parser.add_argument(
|
||||
"--n-bases",
|
||||
type=int,
|
||||
default=-1,
|
||||
help="number of filter weight matrices, default: -1 [use all]",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--n-layers", type=int, default=2, help="number of propagation rounds"
|
||||
)
|
||||
parser.add_argument(
|
||||
"-e",
|
||||
"--n-epochs",
|
||||
type=int,
|
||||
default=50,
|
||||
help="number of training epochs",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-d", "--dataset", type=str, required=True, help="dataset to use"
|
||||
)
|
||||
parser.add_argument("--l2norm", type=float, default=0, help="l2 norm coef")
|
||||
parser.add_argument(
|
||||
"--use-self-loop",
|
||||
default=False,
|
||||
action="store_true",
|
||||
help="include self feature as a special relation",
|
||||
)
|
||||
fp = parser.add_mutually_exclusive_group(required=False)
|
||||
fp.add_argument("--validation", dest="validation", action="store_true")
|
||||
fp.add_argument("--testing", dest="validation", action="store_false")
|
||||
parser.set_defaults(validation=True)
|
||||
|
||||
args = parser.parse_args()
|
||||
print(args)
|
||||
args.bfs_level = args.n_layers + 1 # pruning used nodes for memory
|
||||
main(args)
|
||||
@@ -0,0 +1,59 @@
|
||||
import tensorflow as tf
|
||||
from tensorflow.keras import layers
|
||||
|
||||
|
||||
class BaseRGCN(layers.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
num_nodes,
|
||||
h_dim,
|
||||
out_dim,
|
||||
num_rels,
|
||||
num_bases,
|
||||
num_hidden_layers=1,
|
||||
dropout=0,
|
||||
use_self_loop=False,
|
||||
use_cuda=False,
|
||||
):
|
||||
super(BaseRGCN, self).__init__()
|
||||
self.num_nodes = num_nodes
|
||||
self.h_dim = h_dim
|
||||
self.out_dim = out_dim
|
||||
self.num_rels = num_rels
|
||||
self.num_bases = None if num_bases < 0 else num_bases
|
||||
self.num_hidden_layers = num_hidden_layers
|
||||
self.dropout = dropout
|
||||
self.use_self_loop = use_self_loop
|
||||
self.use_cuda = use_cuda
|
||||
|
||||
# create rgcn layers
|
||||
self.build_model()
|
||||
|
||||
def build_model(self):
|
||||
self.layers = []
|
||||
# i2h
|
||||
i2h = self.build_input_layer()
|
||||
if i2h is not None:
|
||||
self.layers.append(i2h)
|
||||
# h2h
|
||||
for idx in range(self.num_hidden_layers):
|
||||
h2h = self.build_hidden_layer(idx)
|
||||
self.layers.append(h2h)
|
||||
# h2o
|
||||
h2o = self.build_output_layer()
|
||||
if h2o is not None:
|
||||
self.layers.append(h2o)
|
||||
|
||||
def build_input_layer(self):
|
||||
return None
|
||||
|
||||
def build_hidden_layer(self, idx):
|
||||
raise NotImplementedError
|
||||
|
||||
def build_output_layer(self):
|
||||
return None
|
||||
|
||||
def call(self, g, h, r, norm):
|
||||
for layer in self.layers:
|
||||
h = layer(g, h, r, norm)
|
||||
return h
|
||||
@@ -0,0 +1,186 @@
|
||||
"""
|
||||
Utility functions for link prediction
|
||||
Most code is adapted from authors' implementation of RGCN link prediction:
|
||||
https://github.com/MichSchli/RelationPrediction
|
||||
|
||||
"""
|
||||
|
||||
import dgl
|
||||
import numpy as np
|
||||
import tensorflow as tf
|
||||
|
||||
#######################################################################
|
||||
#
|
||||
# Utility function for building training and testing graphs
|
||||
#
|
||||
#######################################################################
|
||||
|
||||
|
||||
def get_adj_and_degrees(num_nodes, triplets):
|
||||
"""Get adjacency list and degrees of the graph"""
|
||||
adj_list = [[] for _ in range(num_nodes)]
|
||||
for i, triplet in enumerate(triplets):
|
||||
adj_list[triplet[0]].append([i, triplet[2]])
|
||||
adj_list[triplet[2]].append([i, triplet[0]])
|
||||
|
||||
degrees = np.array([len(a) for a in adj_list])
|
||||
adj_list = [np.array(a) for a in adj_list]
|
||||
return adj_list, degrees
|
||||
|
||||
|
||||
def sample_edge_neighborhood(adj_list, degrees, n_triplets, sample_size):
|
||||
"""Sample edges by neighborhool expansion.
|
||||
|
||||
This guarantees that the sampled edges form a connected graph, which
|
||||
may help deeper GNNs that require information from more than one hop.
|
||||
"""
|
||||
edges = np.zeros((sample_size), dtype=np.int32)
|
||||
|
||||
# initialize
|
||||
sample_counts = np.array([d for d in degrees])
|
||||
picked = np.array([False for _ in range(n_triplets)])
|
||||
seen = np.array([False for _ in degrees])
|
||||
|
||||
for i in range(0, sample_size):
|
||||
weights = sample_counts * seen
|
||||
|
||||
if np.sum(weights) == 0:
|
||||
weights = np.ones_like(weights)
|
||||
weights[np.where(sample_counts == 0)] = 0
|
||||
|
||||
probabilities = (weights) / np.sum(weights)
|
||||
chosen_vertex = np.random.choice(
|
||||
np.arange(degrees.shape[0]), p=probabilities
|
||||
)
|
||||
chosen_adj_list = adj_list[chosen_vertex]
|
||||
seen[chosen_vertex] = True
|
||||
|
||||
chosen_edge = np.random.choice(np.arange(chosen_adj_list.shape[0]))
|
||||
chosen_edge = chosen_adj_list[chosen_edge]
|
||||
edge_number = chosen_edge[0]
|
||||
|
||||
while picked[edge_number]:
|
||||
chosen_edge = np.random.choice(np.arange(chosen_adj_list.shape[0]))
|
||||
chosen_edge = chosen_adj_list[chosen_edge]
|
||||
edge_number = chosen_edge[0]
|
||||
|
||||
edges[i] = edge_number
|
||||
other_vertex = chosen_edge[1]
|
||||
picked[edge_number] = True
|
||||
sample_counts[chosen_vertex] -= 1
|
||||
sample_counts[other_vertex] -= 1
|
||||
seen[other_vertex] = True
|
||||
|
||||
return edges
|
||||
|
||||
|
||||
def sample_edge_uniform(adj_list, degrees, n_triplets, sample_size):
|
||||
"""Sample edges uniformly from all the edges."""
|
||||
all_edges = np.arange(n_triplets)
|
||||
return np.random.choice(all_edges, sample_size, replace=False)
|
||||
|
||||
|
||||
def generate_sampled_graph_and_labels(
|
||||
triplets,
|
||||
sample_size,
|
||||
split_size,
|
||||
num_rels,
|
||||
adj_list,
|
||||
degrees,
|
||||
negative_rate,
|
||||
sampler="uniform",
|
||||
):
|
||||
"""Get training graph and signals
|
||||
First perform edge neighborhood sampling on graph, then perform negative
|
||||
sampling to generate negative samples
|
||||
"""
|
||||
# perform edge neighbor sampling
|
||||
if sampler == "uniform":
|
||||
edges = sample_edge_uniform(
|
||||
adj_list, degrees, len(triplets), sample_size
|
||||
)
|
||||
elif sampler == "neighbor":
|
||||
edges = sample_edge_neighborhood(
|
||||
adj_list, degrees, len(triplets), sample_size
|
||||
)
|
||||
else:
|
||||
raise ValueError("Sampler type must be either 'uniform' or 'neighbor'.")
|
||||
|
||||
# relabel nodes to have consecutive node ids
|
||||
edges = triplets[edges]
|
||||
src, rel, dst = edges.transpose()
|
||||
uniq_v, edges = np.unique((src, dst), return_inverse=True)
|
||||
src, dst = np.reshape(edges, (2, -1))
|
||||
relabeled_edges = np.stack((src, rel, dst)).transpose()
|
||||
|
||||
# negative sampling
|
||||
samples, labels = negative_sampling(
|
||||
relabeled_edges, len(uniq_v), negative_rate
|
||||
)
|
||||
|
||||
# further split graph, only half of the edges will be used as graph
|
||||
# structure, while the rest half is used as unseen positive samples
|
||||
split_size = int(sample_size * split_size)
|
||||
graph_split_ids = np.random.choice(
|
||||
np.arange(sample_size), size=split_size, replace=False
|
||||
)
|
||||
src = src[graph_split_ids]
|
||||
dst = dst[graph_split_ids]
|
||||
rel = rel[graph_split_ids]
|
||||
|
||||
# build DGL graph
|
||||
print("# sampled nodes: {}".format(len(uniq_v)))
|
||||
print("# sampled edges: {}".format(len(src) * 2))
|
||||
g, rel, norm = build_graph_from_triplets(
|
||||
len(uniq_v), num_rels, (src, rel, dst)
|
||||
)
|
||||
return g, uniq_v, rel, norm, samples, labels
|
||||
|
||||
|
||||
def comp_deg_norm(g):
|
||||
g = g.local_var()
|
||||
in_deg = g.in_degrees(range(g.number_of_nodes())).float().numpy()
|
||||
norm = 1.0 / in_deg
|
||||
norm[np.isinf(norm)] = 0
|
||||
return norm
|
||||
|
||||
|
||||
def build_graph_from_triplets(num_nodes, num_rels, triplets):
|
||||
"""Create a DGL graph. The graph is bidirectional because RGCN authors
|
||||
use reversed relations.
|
||||
This function also generates edge type and normalization factor
|
||||
(reciprocal of node incoming degree)
|
||||
"""
|
||||
g = dgl.DGLGraph()
|
||||
g.add_nodes(num_nodes)
|
||||
src, rel, dst = triplets
|
||||
src, dst = np.concatenate((src, dst)), np.concatenate((dst, src))
|
||||
rel = np.concatenate((rel, rel + num_rels))
|
||||
edges = sorted(zip(dst, src, rel))
|
||||
dst, src, rel = np.array(edges).transpose()
|
||||
g.add_edges(src, dst)
|
||||
norm = comp_deg_norm(g)
|
||||
print("# nodes: {}, # edges: {}".format(num_nodes, len(src)))
|
||||
return g, rel, norm
|
||||
|
||||
|
||||
def build_test_graph(num_nodes, num_rels, edges):
|
||||
src, rel, dst = edges.transpose()
|
||||
print("Test graph:")
|
||||
return build_graph_from_triplets(num_nodes, num_rels, (src, rel, dst))
|
||||
|
||||
|
||||
def negative_sampling(pos_samples, num_entity, negative_rate):
|
||||
size_of_batch = len(pos_samples)
|
||||
num_to_generate = size_of_batch * negative_rate
|
||||
neg_samples = np.tile(pos_samples, (negative_rate, 1))
|
||||
labels = np.zeros(size_of_batch * (negative_rate + 1), dtype=np.float32)
|
||||
labels[:size_of_batch] = 1
|
||||
values = np.random.randint(num_entity, size=num_to_generate)
|
||||
choices = np.random.uniform(size=num_to_generate)
|
||||
subj = choices > 0.5
|
||||
obj = choices <= 0.5
|
||||
neg_samples[subj, 0] = values[subj]
|
||||
neg_samples[obj, 2] = values[obj]
|
||||
|
||||
return np.concatenate((pos_samples, neg_samples)), labels
|
||||
Reference in New Issue
Block a user