-def affinity_clustering(similarities, output, damping=0.5, max_iter=100000, convergence_iter=30, preference_quantile=0.5, random_state=1968):
+def read_similarity_mat(similarities, use_threads=True):
+ df = pd.read_feather(similarities, use_threads=use_threads)
+ mat = np.array(df.drop('_subreddit',1))
+ n = mat.shape[0]
+ mat[range(n),range(n)] = 1
+ return (df._subreddit,mat)
+
+def affinity_clustering(similarities, *args, **kwargs):
+ subreddits, mat = read_similarity_mat(similarities)
+ return _affinity_clustering(mat, subreddits, *args, **kwargs)
+
+def _affinity_clustering(mat, subreddits, output, damping=0.9, max_iter=100000, convergence_iter=30, preference_quantile=0.5, random_state=1968, verbose=True):