X-Git-Url: https://code.communitydata.science/cdsc_reddit.git/blobdiff_plain/811a0d87c4d394c2c7849a613f6aec2d81e49138..07b0dff9bc0dae2ab6f7fb7334007a5269a512ad:/similarities/job_script.sh diff --git a/similarities/job_script.sh b/similarities/job_script.sh index 1158ff0..926b20a 100755 --- a/similarities/job_script.sh +++ b/similarities/job_script.sh @@ -1,4 +1,6 @@ #!/usr/bin/bash +source ~/.bashrc +echo $(hostname) start_spark_cluster.sh -singularity exec /gscratch/comdata/users/nathante/containers/nathante.sif spark-submit --master spark://$(hostname):7077 tfidf.py authors --topN=100000 --inpath=/gscratch/comdata/output/reddit_ngrams/comment_authors.parquet --outpath=/gscratch/scrubbed/comdata/reddit_similarity/tfidf/comment_authors_100k.parquet -singularity exec /gscratch/comdata/users/nathante/containers/nathante.sif stop-all.sh +spark-submit --verbose --master spark://$(hostname):43015 tfidf.py authors --topN=100000 --inpath=../../data/reddit_ngrams/comment_authors_sorted.parquet --outpath=../../data/reddit_similarity/tfidf/comment_authors_100k.parquet +stop-all.sh