-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathbuzzface.py
More file actions
executable file
·71 lines (57 loc) · 2.4 KB
/
Copy pathbuzzface.py
File metadata and controls
executable file
·71 lines (57 loc) · 2.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
#!/bin/env python
import requests
import glob
import re
import os
import xml.etree.ElementTree as ET
from xml.dom import minidom
from lxml import etree
import urllib.parse as urlparse
from bs4 import BeautifulSoup
import html.parser as htmlparser
parser = htmlparser.HTMLParser()
import utils
folder = utils.data_location / 'buzzface'
source_file = folder / 'source' / 'facebook-fact-check.tab'
source_data = utils.read_tsv(source_file)
data = {el['post_id']: {'url': el['Post URL'], 'label': el['Rating']} for el in source_data if el['Post Type'] == 'link'}
# TODO should filter there the interesting classes?
print(len(data))
# download the facebook page
for id, el in data.items():
file_path = folder / 'intermediate' / '{}.html'.format(id)
if not os.path.isfile(file_path):
response = requests.get(el['url'])
utils.write_file_with_path(response.text, folder / 'intermediate', '{}.html'.format(id))
unfiltered = []
results = []
for file_location in glob.glob(str(folder / 'intermediate/*.html')):
#print(file_location)
with open(file_location) as f:
#tree = ET.parse(f)
#soup = BeautifulSoup(f, 'html.parser')
#tree = etree.parse(f, etree.HTMLParser())
str = f.read()
#root = tree.getroot()
#matches = root.findall('a[@tabindex="-1" and target="_blank"]')
#matches = soup.find_all('a', attrs={'tabindex': '-1', 'target': '_blank'})
#matches = tree.xpath('a')
# look for the <a> with tabindex="-1" target="_blank"
fb_urls = re.findall(r'<a\shref="([^>]*)" tabindex="-1" target="_blank"', str)
real_urls = [urlparse.parse_qs(urlparse.urlparse(u).query)['u'] for u in fb_urls]
unique = {u for sublist in real_urls for u in sublist}
#print(unique)
if len(unique) != 1:
print(file_location, unique)
continue
id = file_location.split('/')[-1].split('.')[0]
url = unique.pop()
label = data[id]['label']
label_binary = {'mostly true': 'true', 'mostly false': 'fake'}.get(label, None)
unfiltered.append({'url': url, 'label': label, 'source': 'buzzface'})
if label_binary:
results.append({'url': url, 'label': label_binary, 'source': 'buzzface'})
utils.write_json_with_path(unfiltered, folder / 'intermediate', 'unfiltered.json')
utils.write_json_with_path(results, folder, 'urls.json')
by_domain = utils.compute_by_domain(results)
utils.write_json_with_path(by_domain, folder, 'domains.json')