· 10 years ago · Jul 03, 2016, 10:12 PM
1import requests
2import psycopg2
3import datetime
4import time
5import csv
6import fileinput
7import os
8import json
9
10# This should be stored in /home/ubuntu/
11
12URL = "https://graph.facebook.com/v2.5/posts/"
13MIN_REQUEST_TIME = 3
14ACCESS_TOKEN = "EAAO4ObKSbiIBAH46M9D1h8DpmxAKknCvSqZBSubNdQHVW9TZAouWoZCQsHRlaHhV5aA5g2qOWZCHbxM5n2qCa1UZBTZBH3797sAtAXnQeENVMmAkhSPTQIonph9GUnMvBZCZBUt1O1JeixfZCr0Fxe7zk5j9VlCn66zoZD" # expires august 29
15DB_NAME = "fb"
16DB_USER = "fb"
17DB_PASS = "ZlZi6U6CQzkoKIsA"
18DB_HOST = "localhost"
19CSV_FILE = "fb.csv"
20CSV_FILE_TMP = "fbtmp.csv"
21USERS_PER_REQUEST = 50
22FIELDS = ['id', 'message', 'message_tags', 'name', 'description', 'created_time', 'likes.limit(1).summary(true)', 'link']
23
24conn = psycopg2.connect(database=DB_NAME, user=DB_USER, password=DB_PASS, host=DB_HOST)
25cur = conn.cursor()
26cur.execute("""
27CREATE TABLE IF NOT EXISTS fb_posts (
28 id varchar(64) NOT NULL PRIMARY KEY,
29 username varchar(64) NOT NULL,
30 name text NULL,
31 description text NULL,
32 message text NULL,
33 likes integer NOT NULL,
34 link TEXT NULL,
35 created_time timestamp NOT NULL,
36 grabbed_time timestamp NOT NULL
37);
38
39CREATE TABLE IF NOT EXISTS fb_users (
40 username varchar(64) NOT NULL PRIMARY KEY,
41 grabbed_time timestamp NOT NULL
42);
43""")
44
45
46class FacebookIds:
47 def __init__(self):
48 self.csvfile = None
49 self.datareader = None
50 self.csvfiletmp = None
51 self.csvfiletmp_datawriter = None
52 # init priority usage
53 self._init_csv()
54 self._reset_priority()
55 self._low_priority_done = False
56 self._high_priority_done = False
57 self.end_of_file_reached = False
58
59 def get_id(self):
60 if int(time.strftime("%H")) % 2 == 1:
61 # Odd hour, run low priority
62 priority = 'low'
63 else:
64 # Even hour, run high priority
65 priority = 'high'
66 try:
67 row = self.datareader.next()
68 while row['priority'] != priority:
69 row = self.datareader.next()
70 return row
71 except StopIteration:
72 print('StopIteration')
73 # to process csv files after request we need to set flag
74 self.end_of_file_reached = True
75
76 if priority == 'low':
77 self._low_priority_done = True
78 if priority == 'high':
79 self._high_priority_done = True
80 return None
81
82 def postprocessing(self):
83 if not self.end_of_file_reached:
84 return
85 if self._high_priority_done and self._low_priority_done:
86 self._reset_priority()
87 print "All priorities processed, overriding csv file"
88 self.csvfile.close()
89 self.csvfiletmp.close()
90 os.rename(CSV_FILE_TMP, CSV_FILE)
91 self._init_csv()
92 else:
93 print "Continue from begin of file."
94 self.csvfile.seek(0)
95
96 def _reset_priority(self):
97 self._high_priority_done = False
98 self._low_priority_done = False
99
100 def _init_csv(self):
101 self.csvfile = open(CSV_FILE, "rb")
102 self.datareader = csv.DictReader(self.csvfile, quotechar='"', delimiter=',',
103 quoting=csv.QUOTE_ALL, skipinitialspace=True)
104 try:
105 self.datareader.next() # Hack to read in column names
106 except StopIteration as e:
107 print "Error fb.csv is empty"
108 exit()
109
110 self.csvfiletmp = open(CSV_FILE_TMP, "wb")
111 self.csvfiletmp_datawriter = csv.DictWriter(self.csvfiletmp, fieldnames=['user', 'priority'])
112 self.csvfiletmp_datawriter.writeheader()
113
114 def mark_good(self, rows):
115 print(len(rows), rows)
116 for row in rows:
117 print('adding', row)
118 self.csvfiletmp_datawriter.writerow(row)
119
120
121def get_name_by_id(id):
122 """
123 Here we trying to get username from api. If provided ID was changed - we taking new ID from error message
124 """
125 params = {
126 'access_token': ACCESS_TOKEN,
127 'fields': ','.join(['id', 'username']),
128 }
129 url = "https://graph.facebook.com/v2.5/{0}/".format(id)
130 r = requests.get(url, params=params)
131 response = r.json()
132
133 if r.status_code == 200:
134 return response.get('username')
135
136 elif r.status_code == 400 and response['error']['code'] == 21:
137 # Page moved, remove from CSV and perform a new request.
138 words = response['error']['message'].split()
139 new_id = words[8][:-1]
140 print ("Sleeping for", 3)
141 time.sleep(3)
142 return get_name_by_id(new_id)
143 # Account not found in some reason
144 return None
145
146
147def resolve_migrated_page(request_params):
148 """
149 This method tries to find name that provide next error:
150 'Page ID {old_id} was migrated to page ID {new_id}. Please update your API calls to the new ID'
151 Since we use username instead id, we can't recognize bad name, so here we requesting accounts until we get
152 name that provide this error
153 """
154 names = request_params.get('ids', '').split(',')
155
156 while True:
157 is_last = False
158 if not names:
159 return False, False # returning false if account wasn't found
160 if len(names) == 1:
161 is_last = True
162 part_size = 1
163 request_params['ids'] = ','.join(names[:])
164 print "Bad name found: {0}".format(names[0])
165 else:
166 part_size = len(names)/2
167 request_params['ids'] = ','.join(names[:part_size])
168 print "Looking bad name in {0} accounts".format(len(names))
169 r = requests.get(URL, params=request_params)
170 response = r.json()
171 print ("Sleeping for", 3)
172 time.sleep(3)
173
174 if r.status_code == 200:
175 # we are looking bad account - so skipping 200 response
176 # also we need to take another bunch now
177 names = names[part_size:]
178 continue
179
180 if r.status_code == 400 and response['error']['code'] == 21:
181 if is_last:
182 # if we got right error on 1 user - we found it, need to get new name
183 words = response['error']['message'].split()
184 new_id = words[8][:-1]
185 new_name = get_name_by_id(new_id)
186 return names[0], new_name
187 names = names[:part_size]
188 continue
189
190 elif r.status_code != 200:
191 print("SEARCH FOR NEW NAME PRODUCE UNHANDLED EXCEPTION:")
192 print(r.status_code)
193 print(r.text)
194 continue
195
196
197def remove_user(rows, user):
198 for row in rows:
199 if row['user'] == user:
200 rows.remove(row)
201 return rows
202
203rows = []
204fbids = FacebookIds()
205while True:
206 request_start_time = time.time()
207 start_time = time.time()
208 while len(rows) < USERS_PER_REQUEST:
209 i = fbids.get_id()
210 if not i:
211 # end of file reached
212 break
213 cur.execute("SELECT grabbed_time FROM fb_users WHERE username = %s;", (i['user'],))
214 row = cur.fetchone()
215 if row and time.mktime(row[0].timetuple()) < start_time:
216 start_time = time.mktime(row[0].timetuple())
217 else:
218 start_time = time.time()-86400
219 rows.append(i)
220
221 params = {
222 'access_token': ACCESS_TOKEN,
223 'fields': ','.join(FIELDS),
224 'since': int(start_time), # 1 day before now
225 'limit': 100, # 100 is the limit.
226 'ids': ','.join([row['user'] for row in rows]),
227 }
228
229 r = requests.get(URL, params=params)
230 response = r.json()
231 if r.status_code == 200:
232 print r.status_code
233 print json.dumps(response, indent=4)
234 print json.dumps(params, indent=4)
235
236 request_diff_time = time.time() - request_start_time
237 if request_diff_time < MIN_REQUEST_TIME:
238 print("Sleeping for", MIN_REQUEST_TIME-request_diff_time)
239 time.sleep(MIN_REQUEST_TIME-request_diff_time)
240
241 if r.status_code == 404:
242 msg = response['error']['message']
243 if msg.startswith("(#803) Some of the aliases you requested do not exist"):
244 # Some users are non existant, remove them from the CSV and perform a new request.
245 users = response['error']['message'][55:].split(',')
246 for user in users:
247 print 'removing', user
248 rows = remove_user(rows, user)
249 continue
250 elif msg.startswith("(#803) Cannot query users by their username"):
251 user = response['error']['message'].split()[7][1:-1]
252 rows = remove_user(rows, user)
253 print 'removing', user
254 continue
255 elif r.status_code == 400 and response['error']['code'] == 21:
256 # Page moved, remove from CSV and perform a new request.
257 words = response['error']['message'].split()
258 for row in rows:
259 if row['user'] == words[2]:
260 row['user'] = words[8][:-1]
261 print 'replacing', words[2], 'with', words[8][:-1]
262 break
263 else:
264 print "trying to find migrated id"
265 old_name, new_name = resolve_migrated_page(params)
266 for row in rows:
267 if row['user'] == old_name:
268 row['user'] = new_name
269 print 'replacing', old_name, 'with', new_name
270 break
271 else:
272 print "WARNING: Migrated page not found! Consider to remove problem account: {0}".format(old_name)
273 continue
274 elif r.status_code == 400 and response['error']['code'] == 12:
275 # (#12) singular published story API is deprecated for versions v2.4 and higher
276 # Really not sure why this occurs, all we can do is continue on :(
277 fbids.mark_good(rows)
278 rows = []
279 continue
280 elif r.status_code == 500 and response['error']['code'] == 2 and 'is_transient' in response['error'] and response['error']['is_transient']:
281 # This error seems to be totally random, and the only thing I can see to do is move on.
282 fbids.mark_good(rows)
283 rows = []
284 continue
285 elif r.status_code == 500 and response['error']['code'] == 1:
286 # This error seems to be totally random, and the only thing I can see to do is move on.
287 fbids.mark_good(rows)
288 rows = []
289 continue
290 elif r.status_code == 400 and response['error']['code'] == 100:
291 # (#100) Tried accessing nonexisting field (posts) on node type (URL)
292 fbids.mark_good(rows)
293 rows = []
294 continue
295 elif r.status_code != 200:
296 print("UNHANDLED PROBLEM")
297 print(r.status_code)
298 print(r.text)
299 fbids.mark_good(rows)
300 rows = []
301 continue
302
303 fbids.mark_good(rows)
304 for username, posts in response.items():
305 cur.execute("INSERT INTO fb_users VALUES (%s,%s) ON CONFLICT (username) DO UPDATE SET grabbed_time = %s;", (
306 username,
307 datetime.datetime.now(),
308 datetime.datetime.now()
309 ))
310 for post in posts['data']:
311 cur.execute("INSERT INTO fb_posts VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s) ON CONFLICT DO NOTHING;", (
312 post['id'],
313 username,
314 post['name'] if 'name' in post else None,
315 post['description'] if 'description' in post else None,
316 post['message'] if 'message' in post else None,
317 post['likes']['summary']['total_count'],
318 post['link'] if 'link' in post else None,
319 post['created_time'],
320 datetime.datetime.now()
321 ))
322 conn.commit()
323 rows = []
324 fbids.postprocessing()