· 10 years ago · Jul 03, 2016, 10:15 PM
1import requests
2import psycopg2
3import datetime
4import time
5import csv
6import fileinput
7import os
8import json
9
10# This should be stored in /home/ubuntu/
11
12URL = "https://graph.facebook.com/v2.5/posts/"
13MIN_REQUEST_TIME = 3
14ACCESS_TOKEN = "EAAO4ObKSbiIBAH46M9D1h8DpmxAKknCvSqZBSubNdQHVW9TZAouWoZCQsHRlaHhV5aA5g2qOWZCHbxM5n2qCa1UZBTZBH3797sAtAXnQeENVMmAkhSPTQIonph9GUnMvBZCZBUt1O1JeixfZCr0Fxe7zk5j9VlCn66zoZD" # expires august 29
15DB_NAME = "fb"
16DB_USER = "fb"
17DB_PASS = "ZlZi6U6CQzkoKIsA"
18DB_HOST = "localhost"
19CSV_FILE = "fb.csv"
20CSV_FILE_TMP = "fbtmp.csv"
21USERS_PER_REQUEST = 50
22FIELDS = ['id', 'message', 'message_tags', 'name', 'description', 'created_time', 'likes.limit(1).summary(true)', 'link']
23
24conn = psycopg2.connect(database=DB_NAME, user=DB_USER, password=DB_PASS, host=DB_HOST)
25cur = conn.cursor()
26cur.execute("""
27CREATE TABLE IF NOT EXISTS fb_posts (
28 id varchar(64) NOT NULL PRIMARY KEY,
29 username varchar(64) NOT NULL,
30 name text NULL,
31 description text NULL,
32 message text NULL,
33 likes integer NOT NULL,
34 link TEXT NULL,
35 created_time timestamp NOT NULL,
36 grabbed_time timestamp NOT NULL
37);
38
39CREATE TABLE IF NOT EXISTS fb_users (
40 username varchar(64) NOT NULL PRIMARY KEY,
41 grabbed_time timestamp NOT NULL
42);
43""")
44
45
46class FacebookIds:
47 def __init__(self):
48 self.csvfile = None
49 self.datareader = None
50 self.csvfiletmp = None
51 self.csvfiletmp_datawriter = None
52 # init priority usage
53 self._init_csv()
54 self._reset_priority()
55 self._low_priority_done = False
56 self._high_priority_done = False
57 self.end_of_file_reached = False
58
59 def get_id(self):
60 if int(time.strftime("%H")) % 2 == 1:
61 # Odd hour, run low priority
62 priority = 'low'
63 else:
64 # Even hour, run high priority
65 priority = 'high'
66 try:
67 row = self.datareader.next()
68 while row['priority'] != priority:
69 row = self.datareader.next()
70 return row
71 except StopIteration:
72 print('StopIteration')
73 # to process csv files after request we need to set flag
74 self.end_of_file_reached = True
75
76 if priority == 'low':
77 self._low_priority_done = True
78 if priority == 'high':
79 self._high_priority_done = True
80 return None
81
82 def postprocessing(self):
83 if not self.end_of_file_reached:
84 return
85 # reset flag
86 self.end_of_file_reached = False
87 if self._high_priority_done and self._low_priority_done:
88 self._reset_priority()
89 print "All priorities processed, overriding csv file"
90 self.csvfile.close()
91 self.csvfiletmp.close()
92 os.rename(CSV_FILE_TMP, CSV_FILE)
93 self._init_csv()
94 else:
95 print "Continue from begin of file."
96 self.csvfile.seek(0)
97
98 def _reset_priority(self):
99 self._high_priority_done = False
100 self._low_priority_done = False
101
102 def _init_csv(self):
103 self.csvfile = open(CSV_FILE, "rb")
104 self.datareader = csv.DictReader(self.csvfile, quotechar='"', delimiter=',',
105 quoting=csv.QUOTE_ALL, skipinitialspace=True)
106 try:
107 self.datareader.next() # Hack to read in column names
108 except StopIteration as e:
109 print "Error fb.csv is empty"
110 exit()
111
112 self.csvfiletmp = open(CSV_FILE_TMP, "wb")
113 self.csvfiletmp_datawriter = csv.DictWriter(self.csvfiletmp, fieldnames=['user', 'priority'])
114 self.csvfiletmp_datawriter.writeheader()
115
116 def mark_good(self, rows):
117 print(len(rows), rows)
118 for row in rows:
119 print('adding', row)
120 self.csvfiletmp_datawriter.writerow(row)
121
122
123def get_name_by_id(id):
124 """
125 Here we trying to get username from api. If provided ID was changed - we taking new ID from error message
126 """
127 params = {
128 'access_token': ACCESS_TOKEN,
129 'fields': ','.join(['id', 'username']),
130 }
131 url = "https://graph.facebook.com/v2.5/{0}/".format(id)
132 r = requests.get(url, params=params)
133 response = r.json()
134
135 if r.status_code == 200:
136 return response.get('username')
137
138 elif r.status_code == 400 and response['error']['code'] == 21:
139 # Page moved, remove from CSV and perform a new request.
140 words = response['error']['message'].split()
141 new_id = words[8][:-1]
142 print ("Sleeping for", 3)
143 time.sleep(3)
144 return get_name_by_id(new_id)
145 # Account not found in some reason
146 return None
147
148
149def resolve_migrated_page(request_params):
150 """
151 This method tries to find name that provide next error:
152 'Page ID {old_id} was migrated to page ID {new_id}. Please update your API calls to the new ID'
153 Since we use username instead id, we can't recognize bad name, so here we requesting accounts until we get
154 name that provide this error
155 """
156 names = request_params.get('ids', '').split(',')
157
158 while True:
159 is_last = False
160 if not names:
161 return False, False # returning false if account wasn't found
162 if len(names) == 1:
163 is_last = True
164 part_size = 1
165 request_params['ids'] = ','.join(names[:])
166 print "Bad name found: {0}".format(names[0])
167 else:
168 part_size = len(names)/2
169 request_params['ids'] = ','.join(names[:part_size])
170 print "Looking bad name in {0} accounts".format(len(names))
171 r = requests.get(URL, params=request_params)
172 response = r.json()
173 print ("Sleeping for", 3)
174 time.sleep(3)
175
176 if r.status_code == 200:
177 # we are looking bad account - so skipping 200 response
178 # also we need to take another bunch now
179 names = names[part_size:]
180 continue
181
182 if r.status_code == 400 and response['error']['code'] == 21:
183 if is_last:
184 # if we got right error on 1 user - we found it, need to get new name
185 words = response['error']['message'].split()
186 new_id = words[8][:-1]
187 new_name = get_name_by_id(new_id)
188 return names[0], new_name
189 names = names[:part_size]
190 continue
191
192 elif r.status_code != 200:
193 print("SEARCH FOR NEW NAME PRODUCE UNHANDLED EXCEPTION:")
194 print(r.status_code)
195 print(r.text)
196 continue
197
198
199def remove_user(rows, user):
200 for row in rows:
201 if row['user'] == user:
202 rows.remove(row)
203 return rows
204
205rows = []
206fbids = FacebookIds()
207while True:
208 request_start_time = time.time()
209 start_time = time.time()
210 while len(rows) < USERS_PER_REQUEST:
211 i = fbids.get_id()
212 if not i:
213 # end of file reached
214 break
215 cur.execute("SELECT grabbed_time FROM fb_users WHERE username = %s;", (i['user'],))
216 row = cur.fetchone()
217 if row and time.mktime(row[0].timetuple()) < start_time:
218 start_time = time.mktime(row[0].timetuple())
219 else:
220 start_time = time.time()-86400
221 rows.append(i)
222
223 params = {
224 'access_token': ACCESS_TOKEN,
225 'fields': ','.join(FIELDS),
226 'since': int(start_time), # 1 day before now
227 'limit': 100, # 100 is the limit.
228 'ids': ','.join([row['user'] for row in rows]),
229 }
230
231 r = requests.get(URL, params=params)
232 response = r.json()
233 if r.status_code == 200:
234 print r.status_code
235 print json.dumps(response, indent=4)
236 print json.dumps(params, indent=4)
237
238 request_diff_time = time.time() - request_start_time
239 if request_diff_time < MIN_REQUEST_TIME:
240 print("Sleeping for", MIN_REQUEST_TIME-request_diff_time)
241 time.sleep(MIN_REQUEST_TIME-request_diff_time)
242
243 if r.status_code == 404:
244 msg = response['error']['message']
245 if msg.startswith("(#803) Some of the aliases you requested do not exist"):
246 # Some users are non existant, remove them from the CSV and perform a new request.
247 users = response['error']['message'][55:].split(',')
248 for user in users:
249 print 'removing', user
250 rows = remove_user(rows, user)
251 continue
252 elif msg.startswith("(#803) Cannot query users by their username"):
253 user = response['error']['message'].split()[7][1:-1]
254 rows = remove_user(rows, user)
255 print 'removing', user
256 continue
257 elif r.status_code == 400 and response['error']['code'] == 21:
258 # Page moved, remove from CSV and perform a new request.
259 words = response['error']['message'].split()
260 for row in rows:
261 if row['user'] == words[2]:
262 row['user'] = words[8][:-1]
263 print 'replacing', words[2], 'with', words[8][:-1]
264 break
265 else:
266 print "trying to find migrated id"
267 old_name, new_name = resolve_migrated_page(params)
268 for row in rows:
269 if row['user'] == old_name:
270 row['user'] = new_name
271 print 'replacing', old_name, 'with', new_name
272 break
273 else:
274 print "WARNING: Migrated page not found! Consider to remove problem account: {0}".format(old_name)
275 continue
276 elif r.status_code == 400 and response['error']['code'] == 12:
277 # (#12) singular published story API is deprecated for versions v2.4 and higher
278 # Really not sure why this occurs, all we can do is continue on :(
279 fbids.mark_good(rows)
280 rows = []
281 continue
282 elif r.status_code == 500 and response['error']['code'] == 2 and 'is_transient' in response['error'] and response['error']['is_transient']:
283 # This error seems to be totally random, and the only thing I can see to do is move on.
284 fbids.mark_good(rows)
285 rows = []
286 continue
287 elif r.status_code == 500 and response['error']['code'] == 1:
288 # This error seems to be totally random, and the only thing I can see to do is move on.
289 fbids.mark_good(rows)
290 rows = []
291 continue
292 elif r.status_code == 400 and response['error']['code'] == 100:
293 # (#100) Tried accessing nonexisting field (posts) on node type (URL)
294 fbids.mark_good(rows)
295 rows = []
296 continue
297 elif r.status_code != 200:
298 print("UNHANDLED PROBLEM")
299 print(r.status_code)
300 print(r.text)
301 fbids.mark_good(rows)
302 rows = []
303 continue
304
305 fbids.mark_good(rows)
306 for username, posts in response.items():
307 cur.execute("INSERT INTO fb_users VALUES (%s,%s) ON CONFLICT (username) DO UPDATE SET grabbed_time = %s;", (
308 username,
309 datetime.datetime.now(),
310 datetime.datetime.now()
311 ))
312 for post in posts['data']:
313 cur.execute("INSERT INTO fb_posts VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s) ON CONFLICT DO NOTHING;", (
314 post['id'],
315 username,
316 post['name'] if 'name' in post else None,
317 post['description'] if 'description' in post else None,
318 post['message'] if 'message' in post else None,
319 post['likes']['summary']['total_count'],
320 post['link'] if 'link' in post else None,
321 post['created_time'],
322 datetime.datetime.now()
323 ))
324 conn.commit()
325 rows = []
326 fbids.postprocessing()