X-Git-Url: https://code.communitydata.science/cdsc_reddit.git/blobdiff_plain/6e43294a41e030e557d7e612f1e6ddb063482689..197518a222a321a8027c3dc5a4121350c47d0779:/similarities/tfidf.py?ds=inline diff --git a/similarities/tfidf.py b/similarities/tfidf.py index 94dcbf5..bbae528 100644 --- a/similarities/tfidf.py +++ b/similarities/tfidf.py @@ -26,11 +26,12 @@ def tfidf(inpath, outpath, topN, term_colname, exclude, included_subreddits): def tfidf_weekly(inpath, outpath, topN, term_colname, exclude, included_subreddits): return _tfidf_wrapper(build_weekly_tfidf_dataset, inpath, outpath, topN, term_colname, exclude, included_subreddits) -def tfidf_authors(outpath='/gscratch/comdata/output/reddit_similarity/tfidf/comment_authors.parquet', +def tfidf_authors(inpath="/gscratch/comdata/output/reddit_ngrams/comment_authors.parquet", + outpath='/gscratch/comdata/output/reddit_similarity/tfidf/comment_authors.parquet', topN=None, included_subreddits=None): - return tfidf("/gscratch/comdata/output/reddit_ngrams/comment_authors.parquet", + return tfidf(inpath, outpath, topN, 'author', @@ -38,11 +39,12 @@ def tfidf_authors(outpath='/gscratch/comdata/output/reddit_similarity/tfidf/comm included_subreddits=included_subreddits ) -def tfidf_terms(outpath='/gscratch/comdata/output/reddit_similarity/tfidf/comment_terms.parquet', +def tfidf_terms(inpath="/gscratch/comdata/output/reddit_ngrams/comment_terms.parquet", + outpath='/gscratch/comdata/output/reddit_similarity/tfidf/comment_terms.parquet', topN=None, included_subreddits=None): - return tfidf("/gscratch/comdata/output/reddit_ngrams/comment_terms.parquet", + return tfidf(inpath, outpath, topN, 'term', @@ -50,11 +52,12 @@ def tfidf_terms(outpath='/gscratch/comdata/output/reddit_similarity/tfidf/commen included_subreddits=included_subreddits ) -def tfidf_authors_weekly(outpath='/gscratch/comdata/output/reddit_similarity/tfidf_weekly/comment_authors.parquet', +def tfidf_authors_weekly(inpath="/gscratch/comdata/output/reddit_ngrams/comment_authors.parquet", + outpath='/gscratch/comdata/output/reddit_similarity/tfidf_weekly/comment_authors.parquet', topN=None, - include_subreddits=None): + included_subreddits=None): - return tfidf_weekly("/gscratch/comdata/output/reddit_ngrams/comment_authors.parquet", + return tfidf_weekly(inpath, outpath, topN, 'author', @@ -62,16 +65,18 @@ def tfidf_authors_weekly(outpath='/gscratch/comdata/output/reddit_similarity/tfi included_subreddits=included_subreddits ) -def tfidf_terms_weekly(outpath='/gscratch/comdata/output/reddit_similarity/tfidf_weekly/comment_terms.parquet', - topN=25000): +def tfidf_terms_weekly(inpath="/gscratch/comdata/output/reddit_ngrams/comment_terms.parquet", + outpath='/gscratch/comdata/output/reddit_similarity/tfidf_weekly/comment_terms.parquet', + topN=None, + included_subreddits=None): - return tfidf_weekly("/gscratch/comdata/output/reddit_ngrams/comment_terms.parquet", + return tfidf_weekly(inpath, outpath, topN, 'term', [], - included_subreddits=None + included_subreddits=included_subreddits )