Make domains configurable for hc score dsa1-score
shawnmjones opened this issue · comments
Shawn M. Jones commented
Scoring mementos by Padia's work depends upon a list of domains for the categories of news, image sharing sites, video sharing sites, blog sites, and social media sites. These lists will likely change over time and should be configurable by the end user.
The scoring is handled on lines 431 - 451 of hypercane/score/dsa1_ranking
. We will have to make it generic.
hypercane/hypercane/score/dsa1_ranking.py
Lines 431 to 451 in 2e071d7
if domain in blog_sources: | |
return 0.4 | |
elif domain in wikipedia_imagesharing_sources: | |
return 0.6 | |
elif domain.upper() in pew_news_sources: | |
return 0.7 | |
elif domain in w3newspapers_sources: | |
return 0.7 | |
elif 'news' in domain: | |
return 0.7 | |
elif domain in wikipedia_video_sources: | |
return 0.7 | |
elif domain in adobe_socialmedia_sources: | |
return 0.5 | |
The lists themselves are on lines 15 - 386 of that same file.
hypercane/hypercane/score/dsa1_ranking.py
Lines 15 to 386 in 2e071d7
blog_sources = [ | |
'blogger.com', | |
'blogspot.com', | |
'wordpress.com', | |
'typepad.com' | |
] | |
# image sharing websites as per https://en.wikipedia.org/wiki/List_of_image-sharing_websites | |
wikipedia_imagesharing_sources = [ | |
'500px.com', | |
'album2.com', | |
'bilddagboken.se', | |
'myphotodiary.com', | |
'kuvapaivakirja.fi', | |
'bildedagboka.no', | |
'billeddagbog.dk', | |
'deviantart.com', | |
'dronestagr.am', | |
'flickr.com', | |
'fotki.com', | |
'fotolog.com', | |
'fotolog.net', | |
'geograph.org.uk', | |
'photos.google.com', | |
'instagram.com', | |
'imgur.com', | |
'ipernity.com', | |
'jalbum.net', | |
'photobucket.com', | |
'pinterest.com', | |
'pixabay.com', | |
'securetribeapp.com', | |
'shutterflyinc.com', | |
'smugmug.com', | |
'snapfish.com', | |
'unsplash.com' | |
] | |
# video domains as per https://en.wikipedia.org/wiki/List_of_video_hosting_services#Specifically_dedicated_video_hosting_websites | |
wikipedia_video_sources = [ | |
'acfun.cn', | |
'afreecatv.com', | |
'aparat.com', | |
'bigo.tv', | |
'bilibili.com', | |
'bitchute.com', | |
'dailymotion.com', | |
'godtube.com', | |
'iqiyi.com', | |
'liveleak.com', | |
'metacafe.com', | |
'mixer.com', | |
'nicovideo.jp', | |
'periscope.tv', | |
'rutube.ru', | |
'schooltube.com', | |
'smashcast.tv', | |
'trilulilu.ro', | |
'tudou.com', | |
'tune.pk', | |
'twitch.tv', | |
'vbox7.com', | |
'veoh.com', | |
'vimeo.com', | |
'youku.com', | |
'younow.com', | |
'youtube.com' | |
] | |
# social media domains as per https://helpx.adobe.com/analytics/kb/list-social-networks.html | |
adobe_socialmedia_sources = [ | |
'12seconds.tv', | |
'4travel.jp', | |
'advogato.org', | |
'ameba.jp', | |
'anobii.com', | |
'answers.yahoo.com', | |
'asmallworld.net', | |
'avforums.com', | |
'backtype.com', | |
'badoo.com', | |
'bebo.com', | |
'bigadda.com', | |
'bigtent.com', | |
'biip.no', | |
'blackplanet.com', | |
'blog.seesaa.jp', | |
'blogspot.com', | |
'blogster.com', | |
'blomotion.jp', | |
'bolt.com', | |
'brightkite.com', | |
'buzznet.com', | |
'cafemom.com', | |
'care2.com', | |
'classmates.com', | |
'cloob.com', | |
'collegeblender.com', | |
'cyworld.co.kr', | |
'cyworld.com.cn', | |
'dailymotion.com', | |
'delicious.com', | |
'deviantart.com', | |
'digg.com', | |
'diigo.com', | |
'disqus.com', | |
'draugiem.lv', | |
'facebook.com', | |
'faceparty.com', | |
'fc2.com', | |
'flickr.com', | |
'flixster.com', | |
'fotolog.com', | |
'foursquare.com', | |
'friendfeed.com', | |
'friendsreunited.co.uk', | |
'friendsreunited.com', | |
'friendster.com', | |
'fubar.com', | |
'gaiaonline.com', | |
'geni.com', | |
'goodreads.com', | |
'grono.net', | |
'habbo.com', | |
'hatena.ne.jp', | |
'hi5.com', | |
'hotnews.infoseek.co.jp', | |
'hyves.nl', | |
'ibibo.com', | |
'identi.ca', | |
'imeem.com', | |
'instagram.com', | |
'intensedebate.com', | |
'irc-galleria.net', | |
'iwiw.hu', | |
'jaiku.com', | |
'jp.myspace.com', | |
'kaixin001.com', | |
'kaixin002.com', | |
'kakaku.com', | |
'kanshin.com', | |
'kozocom.com', | |
'last.fm', | |
'linkedin.com', | |
'livejournal.com', | |
'lnkd.in', | |
'matome.naver.jp', | |
'me2day.net', | |
'meetup.com', | |
'mister-wong.com', | |
'mixi.jp', | |
'mixx.com', | |
'mouthshut.com', | |
'mp.weixin.qq.com', | |
'multiply.com', | |
'mumsnet.com', | |
'myheritage.com', | |
'mylife.com', | |
'myspace.com', | |
'myyearbook.com', | |
'nasza-klasa.pl', | |
'netlog.com', | |
'nettby.no', | |
'netvibes.com', | |
'nextdoor.com', | |
'nicovideo.jp', | |
'ning.com', | |
'odnoklassniki.ru', | |
'ok.ru', | |
'orkut.com', | |
'pakila.jp', | |
'photobucket.com', | |
'pinterest.at', | |
'pinterest.be', | |
'pinterest.ca', | |
'pinterest.ch', | |
'pinterest.cl', | |
'pinterest.co', | |
'pinterest.co.kr', | |
'pinterest.co.uk', | |
'pinterest.com', | |
'pinterest.de', | |
'pinterest.dk', | |
'pinterest.es', | |
'pinterest.fr', | |
'pinterest.hu', | |
'pinterest.ie', | |
'pinterest.in', | |
'pinterest.jp', | |
'pinterest.nz', | |
'pinterest.ph', | |
'pinterest.pt', | |
'pinterest.se', | |
'plaxo.com', | |
'plurk.com', | |
'plus.google.com', | |
'plus.url.google.com', | |
'po.st', | |
'reddit.com', | |
'renren.com', | |
'skyrock.com', | |
'slideshare.net', | |
'smcb.jp', | |
'smugmug.com', | |
'sonico.com', | |
'studivz.net', | |
'stumbleupon.com', | |
't.163.com', | |
't.co', | |
't.hexun.com', | |
't.ifeng.com', | |
't.people.com.cn', | |
't.qq.com', | |
't.sina.com.cn', | |
't.sohu.com', | |
'tabelog.com', | |
'tagged.com', | |
'taringa.net', | |
'thefancy.com', | |
'toutiao.com', | |
'tripit.com', | |
'trombi.com', | |
'trytrend.jp', | |
'tuenti.com', | |
'tumblr.com', | |
'twine.com', | |
'twitter.com', | |
'uhuru.jp', | |
'viadeo.com', | |
'vimeo.com', | |
'vk.com', | |
'wayn.com', | |
'weibo.com', | |
'weourfamily.com', | |
'wer-kennt-wen.de', | |
'wordpress.com', | |
'xanga.com', | |
'xing.com', | |
'yammer.com', | |
'yaplog.jp', | |
'yelp.co.uk', | |
'yelp.com', | |
'youku.com', | |
'youtube.com', | |
'yozm.daum.net', | |
'yuku.com', | |
'zhihu.com', | |
'zooomr.com' | |
] | |
# news domains from https://pewresearch-org-preprod.go-vip.co/journalism/2019/07/23/state-of-the-news-media-methodology/#digital-native-news-outlet-audit | |
pew_news_sources = [ | |
'12UP.COM', | |
'247SPORTS.COM', | |
'90MIN.COM', | |
'APLUS.COM', | |
'BGR.COM', | |
'BLEACHERREPORT.COM', | |
'BREITBART.COM', | |
'BUSINESSINSIDER.COM', | |
'BUSTLE.COM', | |
'BUZZFEED.COM', | |
'BUZZFEEDNEWS.COM', | |
'CHEATSHEET.COM', | |
'CINEMABLEND.COM', | |
'CNET.COM', | |
'COMICBOOK.COM', | |
'DAILYDOT.COM', | |
'DEADSPIN.COM', | |
'DIGITALTRENDS.COM', | |
'EATER.COM', | |
'ELITEDAILY.COM', | |
'ENGADGET.COM', | |
'FIVETHIRTYEIGHT.COM', | |
'GAMESPOT.COM', | |
'GIZMODO.COM', | |
'HELLOGIGGLES.COM', | |
'HOLLYWOODLIFE.COM', | |
'HUFFINGTONPOST.COM', | |
'IBTIMES.COM', | |
'IFLSCIENCE.COM', | |
'IGN.COM', | |
'IJR.COM', | |
'IJREVIEW.COM', | |
'INVESTOPEDIA.COM', | |
'JEZEBEL.COM', | |
'MARKETWATCH.COM', | |
'MASHABLE.COM', | |
'MAXPREPS.COM', | |
'MIC.COM', | |
'OPPOSINGVIEWS.COM', | |
'POLITICO.COM', | |
'POLYGON.COM', | |
'QZ.COM', | |
'RARE.US', | |
'RAWSTORY.COM', | |
'REFINERY29.COM', | |
'SALON.COM', | |
'SBNATION.COM', | |
'SLATE.COM', | |
'TECHRADAR.COM', | |
'THEBLAZE.COM', | |
'THEDAILYBEAST.COM', | |
'THEROOT.COM', | |
'THEVERGE.COM', | |
'THISISINSIDER.COM', | |
'THRILLIST.COM', | |
'TMZ.COM', | |
'TOPIX.COM', | |
'TOPIX.NET', | |
'UPROXX.COM', | |
'UPWORTHY.COM', | |
'VOX.COM' | |
] | |
# sources from https://www.w3newspapers.com/newssites/ | |
w3newspapers_sources = [ | |
'aljazeera.com', | |
'nytimes.com', | |
'wsj.com', | |
'huffpost.com', | |
'washingtonpost.com', | |
'latimes.com', | |
'reuters.com', | |
'abcnews.go.com', | |
'usatoday.com', | |
'bloomberg.com', | |
'nbcnews.com', | |
'dailymail.co.uk', | |
'theguardian.com', | |
'thesun.co.uk', | |
'mirror.co.uk', | |
'telegraph.co.uk', | |
'bbc.com', | |
'thestar.com', | |
'theglobeandmail.com', | |
'news.com.au', | |
'forbes.com', | |
'cnbc.com', | |
'chinadaily.com.cn', | |
'chron.com', | |
'nypost.com', | |
'usnews.com', | |
'dw.com', | |
'indiatimes.com', | |
'thehindu.com', | |
'indianexpress.com', | |
'hindustantimes.com', | |
'cbsnews.com', | |
'time.com', | |
'sfgate.com', | |
'thehill.com', | |
'thedailybeast.com', | |
'newsweek.com', | |
'theatlantic.com', | |
'nzherald.co.nz', | |
'herald.co.zw', | |
'vanguardngr.com', | |
'dailysun.co.za', | |
'thejakartapost.com', | |
'thestar.com.my', | |
'straitstimes.com', | |
'bangkokpost.com', | |
'japantimes.co.jp', | |
'thedailystar.net', | |
'dawn.com', | |
'alarabiya.net', | |
'hollywoodreporter.com', | |
'scmp.com', | |
'aljazeera.com', | |
'voanews.com' | |
] |
Shawn M. Jones commented
This is a duplicate of #10. Closing.