mirror of
https://github.com/timberjoegithub/GoogleScrape.git
synced 2026-07-22 00:19:48 +00:00
Merge branch 'main' of https://github.com/timberjoegithub/GoogleScrape
This commit is contained in:
@@ -21,3 +21,4 @@ social.py
|
|||||||
.vscode/
|
.vscode/
|
||||||
googleinfo.py
|
googleinfo.py
|
||||||
social.examples.py
|
social.examples.py
|
||||||
|
debug.log
|
||||||
|
|||||||
+19
-13
@@ -1,13 +1,19 @@
|
|||||||
autopep8==1.5.7
|
autopep8
|
||||||
et-xmlfile==1.1.0
|
et-xmlfile
|
||||||
numpy==1.21.2
|
numpy
|
||||||
openpyxl==3.0.9
|
openpyxl
|
||||||
pandas==1.3.3
|
pandas
|
||||||
pkg_resources==0.0.0
|
pycodestyle
|
||||||
pycodestyle==2.7.0
|
python-dateutil
|
||||||
python-dateutil==2.8.2
|
pytz
|
||||||
pytz==2021.1
|
selenium
|
||||||
selenium==3.141.0
|
six
|
||||||
six==1.16.0
|
toml
|
||||||
toml==0.10.2
|
urllib3
|
||||||
urllib3==1.26.6
|
aiohttp
|
||||||
|
instagrapi
|
||||||
|
jsonpickle
|
||||||
|
tweepy
|
||||||
|
sqlalchemy
|
||||||
|
googlemaps
|
||||||
|
moviepy
|
||||||
@@ -117,15 +117,32 @@ class Posts(Base):
|
|||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def preload():
|
def preload():
|
||||||
|
"""
|
||||||
|
Removes a specific file if it exists.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
None
|
||||||
|
"""
|
||||||
|
|
||||||
file=pathlib.Path("./config/joeteststeele_uuid_and_cookie.json")
|
file=pathlib.Path("./config/joeteststeele_uuid_and_cookie.json")
|
||||||
if pathlib.Path.exists(file):
|
if pathlib.Path.exists(file):
|
||||||
pathlib.Path.unlink(file)
|
pathlib.Path.unlink(file)
|
||||||
today = datetime.today().strftime('%Y-%m-%d')
|
# today = datetime.today().strftime('%Y-%m-%d')
|
||||||
return
|
return
|
||||||
|
|
||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def clearlist(my_list):
|
def clearlist(my_list):
|
||||||
|
"""
|
||||||
|
Clears all elements in a list.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
my_list (list): The list to be cleared.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list: The input list with all elements cleared.
|
||||||
|
"""
|
||||||
|
|
||||||
for listelement in my_list:
|
for listelement in my_list:
|
||||||
listelement.clear
|
listelement.clear
|
||||||
return my_list
|
return my_list
|
||||||
@@ -231,19 +248,31 @@ def get_twitter_conn_v2(api_key, api_secret, access_token, access_token_secret)
|
|||||||
|
|
||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def get_hastags (address, name, type):
|
def get_hastags(address, name, hashtype):
|
||||||
|
"""
|
||||||
|
Generates hashtags based on the address, name, and type provided.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
address (str): The address related to the content.
|
||||||
|
name (str): The name associated with the content.
|
||||||
|
hashtype (str): The type of hashtags to generate.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The generated hashtags based on the input parameters.
|
||||||
|
"""
|
||||||
|
|
||||||
nameNoSpaces = re.sub( r'[^a-zA-Z]','',name)
|
nameNoSpaces = re.sub( r'[^a-zA-Z]','',name)
|
||||||
addressdict = address.rsplit(r' ',3)
|
addressdict = address.rsplit(r' ',3)
|
||||||
zip = addressdict[3]
|
zip_code = addressdict[3]
|
||||||
state = addressdict[2]
|
state = addressdict[2]
|
||||||
city = re.sub( r'[^a-zA-Z]','',addressdict[1])
|
city = re.sub( r'[^a-zA-Z]','',addressdict[1])
|
||||||
if 'short' in type:
|
if 'short' in hashtype:
|
||||||
defaulttags = '#'+nameNoSpaces+' #foodie #food #joeeatswhat @timberjoe'
|
defaulttags = '#'+nameNoSpaces+' #foodie #food #joeeatswhat @timberjoe'
|
||||||
else:
|
else:
|
||||||
defaulttags = "\n\n\n#"+nameNoSpaces+" #foodie #music #food #travel #drinks #instagood #feedme #joeeatswhat @timberjoe"
|
defaulttags = "\n\n\n#"+nameNoSpaces+" #foodie #music #food #travel #drinks #instagood #feedme #joeeatswhat @timberjoe"
|
||||||
citytag = "#"+city
|
citytag = "#"+city
|
||||||
statetag = "#"+state
|
statetag = "#"+state
|
||||||
ziptag = "#"+zip
|
ziptag = "#"+zip_code
|
||||||
if statetag == 'FL':
|
if statetag == 'FL':
|
||||||
statetag += ' #Florida'
|
statetag += ' #Florida'
|
||||||
fulltag = defaulttags+" "+citytag+" "+statetag+" "+ziptag
|
fulltag = defaulttags+" "+citytag+" "+statetag+" "+ziptag
|
||||||
@@ -255,6 +284,16 @@ def get_hastags (address, name, type):
|
|||||||
|
|
||||||
# Grab a count of how far we need to scroll
|
# Grab a count of how far we need to scroll
|
||||||
def counter_google(driver):
|
def counter_google(driver):
|
||||||
|
"""
|
||||||
|
Counts the number of Google search results pages.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
driver: The Selenium WebDriver instance.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: The total number of search result pages.
|
||||||
|
"""
|
||||||
|
|
||||||
result = driver.find_element(By.CLASS_NAME,'Qha3nb').text
|
result = driver.find_element(By.CLASS_NAME,'Qha3nb').text
|
||||||
result = result.replace(',', '')
|
result = result.replace(',', '')
|
||||||
result = result.split(' ')
|
result = result.split(' ')
|
||||||
@@ -264,6 +303,16 @@ def counter_google(driver):
|
|||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def make_montage_video_from_google(inphotos):
|
def make_montage_video_from_google(inphotos):
|
||||||
|
"""
|
||||||
|
Creates a montage video from a list of input photos.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
inphotos (list): List of paths to input photo files.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple: A tuple containing the path to the output video file and a boolean indicating success.
|
||||||
|
"""
|
||||||
|
|
||||||
# Load the photos from the folder
|
# Load the photos from the folder
|
||||||
# Set the duration of each photo to 2 seconds
|
# Set the duration of each photo to 2 seconds
|
||||||
if inphotos:
|
if inphotos:
|
||||||
@@ -296,6 +345,12 @@ def make_montage_video_from_google(inphotos):
|
|||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def is_docker():
|
def is_docker():
|
||||||
|
"""
|
||||||
|
Checks if the code is running in a Docker container.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True if running in a Docker container, False otherwise.
|
||||||
|
"""
|
||||||
cgroup = Path('/proc/self/cgroup')
|
cgroup = Path('/proc/self/cgroup')
|
||||||
#print (cgroup.read_text())
|
#print (cgroup.read_text())
|
||||||
return Path('/.dockerenv').is_file() or cgroup.is_file() and 'docker' in cgroup.read_text()
|
return Path('/.dockerenv').is_file() or cgroup.is_file() and 'docker' in cgroup.read_text()
|
||||||
@@ -303,13 +358,27 @@ def is_docker():
|
|||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def post_facebook_video(group_id, video_path, auth_token, title, content, date, rating, address):
|
def post_facebook_video(group_id, video_path, auth_token, title, content, date, rating, address):
|
||||||
|
"""
|
||||||
|
Posts a video to a Facebook group with specified details.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
group_id (str): The ID of the Facebook group.
|
||||||
|
video_path (list): List of paths to the video files to be uploaded.
|
||||||
|
auth_token (str): The authentication token for posting to Facebook.
|
||||||
|
title (str): The title of the video.
|
||||||
|
content (str): Additional content to be included in the post.
|
||||||
|
date (str): The date of the post.
|
||||||
|
rating (str): The rating associated with the post.
|
||||||
|
address (str): The address related to the post.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict or bool: The response JSON if successful, False if an error occurs.
|
||||||
|
"""
|
||||||
|
|
||||||
url = f"https://graph-video.facebook.com/{group_id}/videos?access_token=" + auth_token
|
url = f"https://graph-video.facebook.com/{group_id}/videos?access_token=" + auth_token
|
||||||
files={}
|
files={}
|
||||||
addresshtml = re.sub(" ", ".",address)
|
addresshtml = re.sub(" ", ".",address)
|
||||||
#args={}
|
|
||||||
#data["message"]=title + "\n"+address+"\n\n"+ content + "\n"+rating+"\n"+date
|
|
||||||
for eachfile in video_path:
|
for eachfile in video_path:
|
||||||
# my_dict['key'].append(1)
|
|
||||||
files.update({eachfile: open(eachfile, 'rb')})
|
files.update({eachfile: open(eachfile, 'rb')})
|
||||||
data = { "title":title,"description" : title + "\n"+ address+"\nGoogle map to destination: "
|
data = { "title":title,"description" : title + "\n"+ address+"\nGoogle map to destination: "
|
||||||
r"https://www.google.com/maps/dir/?api=1&destination="+addresshtml +"\n\n"+ content +
|
r"https://www.google.com/maps/dir/?api=1&destination="+addresshtml +"\n\n"+ content +
|
||||||
@@ -373,8 +442,8 @@ def get_google_data(driver,outputs ):
|
|||||||
# > span.kvMYJc
|
# > span.kvMYJc
|
||||||
except Exception as error:
|
except Exception as error:
|
||||||
score = "Unknown"
|
score = "Unknown"
|
||||||
|
print ('Error: ',error)
|
||||||
more_specific_pics = data.find_elements(By.CLASS_NAME, 'Tya61d')
|
more_specific_pics = data.find_elements(By.CLASS_NAME, 'Tya61d')
|
||||||
|
|
||||||
# Grab more info from google maps entry on this particular review
|
# Grab more info from google maps entry on this particular review
|
||||||
if outputs['postssession'].query(Posts).filter(Posts.name == name,Posts.google != 1) or env.forcegoogleupdate:
|
if outputs['postssession'].query(Posts).filter(Posts.name == name,Posts.google != 1) or env.forcegoogleupdate:
|
||||||
gmaps = googlemaps.Client(env.googleapipass)
|
gmaps = googlemaps.Client(env.googleapipass)
|
||||||
@@ -386,11 +455,11 @@ def get_google_data(driver,outputs ):
|
|||||||
# Get place details
|
# Get place details
|
||||||
# googledetails = gmaps.place(place_id)
|
# googledetails = gmaps.place(place_id)
|
||||||
try:
|
try:
|
||||||
businessurl = (details['result']['website'])
|
businessurl = details['result']['website']
|
||||||
latitude = (details['result']['geometry']['location']['lat'])
|
latitude = details['result']['geometry']['location']['lat']
|
||||||
longitude = (details['result']['geometry']['location']['lng'])
|
longitude = details['result']['geometry']['location']['lng']
|
||||||
pluscode = (details['result']['plus_code']['compound_code'])
|
pluscode = details['result']['plus_code']['compound_code']
|
||||||
googleurl = (details['result']['url'])
|
googleurl = details['result']['url']
|
||||||
database_update_row(name,"businessurl",businessurl,"onlyempty",outputs)
|
database_update_row(name,"businessurl",businessurl,"onlyempty",outputs)
|
||||||
database_update_row(name,"latitude",latitude,"onlyempty",outputs)
|
database_update_row(name,"latitude",latitude,"onlyempty",outputs)
|
||||||
database_update_row(name,"longitude",longitude,"onlyempty",outputs)
|
database_update_row(name,"longitude",longitude,"onlyempty",outputs)
|
||||||
@@ -471,6 +540,19 @@ def get_google_data(driver,outputs ):
|
|||||||
|
|
||||||
# Do the google_scroll
|
# Do the google_scroll
|
||||||
def google_scroll(counter_google,driver):
|
def google_scroll(counter_google,driver):
|
||||||
|
"""
|
||||||
|
Scrolls down a Google search results page a specified number of times.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
counter_google (int): The number of times to scroll down the page.
|
||||||
|
driver: The Selenium WebDriver instance.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: The result of the last scroll operation.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
Exception: If an error occurs during scrolling.
|
||||||
|
"""
|
||||||
print('google_scroll...')
|
print('google_scroll...')
|
||||||
time.sleep(3)
|
time.sleep(3)
|
||||||
scrollable_div = driver.find_element(By.XPATH,
|
scrollable_div = driver.find_element(By.XPATH,
|
||||||
@@ -595,11 +677,11 @@ def write_to_database(data, outputs):
|
|||||||
|
|
||||||
def database_update_row(review_name,column_name,column_value,update_style,outputs):
|
def database_update_row(review_name,column_name,column_value,update_style,outputs):
|
||||||
try:
|
try:
|
||||||
if update_style == "forceall":
|
if update_style == "forceall" and column_value != False:
|
||||||
outputs['postssession'].query(Posts).filter(Posts.name == review_name).update\
|
outputs['postssession'].query(Posts).filter(Posts.name == review_name).update\
|
||||||
({column_name : column_value})
|
({column_name : column_value})
|
||||||
print (' Force Updated ',column_name, ' to: ',column_value)
|
print (' Force Updated ',column_name, ' to: ',column_value)
|
||||||
elif update_style == "onlyempty":
|
elif update_style == "onlyempty" and column_value != False:
|
||||||
postval = outputs['postssession'].query(Posts).filter(Posts.name == review_name,\
|
postval = outputs['postssession'].query(Posts).filter(Posts.name == review_name,\
|
||||||
getattr(Posts,column_name).is_not(null())).all()
|
getattr(Posts,column_name).is_not(null())).all()
|
||||||
if len(postval) == 0 :
|
if len(postval) == 0 :
|
||||||
@@ -609,10 +691,10 @@ def database_update_row(review_name,column_name,column_value,update_style,output
|
|||||||
elif update_style == "toggletrue":
|
elif update_style == "toggletrue":
|
||||||
postval = outputs['postssession'].query(Posts).filter(Posts.name == review_name,\
|
postval = outputs['postssession'].query(Posts).filter(Posts.name == review_name,\
|
||||||
getattr(Posts,column_name).is_not(1)).all()
|
getattr(Posts,column_name).is_not(1)).all()
|
||||||
if len(postval) == 0 :
|
|
||||||
outputs['postssession'].query(Posts).filter(Posts.name == review_name).update\
|
outputs['postssession'].query(Posts).filter(Posts.name == review_name).update\
|
||||||
({column_name : column_value})
|
({column_name : "1"})
|
||||||
print (' Updated ',column_name, ' on value: ',postval[0].column_value, ' to: ',column_value)
|
print (' Updated ',column_name, ' on value: ',postval[0].column_value, ' to: ',\
|
||||||
|
column_value)
|
||||||
except Exception as error:
|
except Exception as error:
|
||||||
print(" Not able to write to post data table to update ",review_name," ",column_name,"\
|
print(" Not able to write to post data table to update ",review_name," ",column_name,"\
|
||||||
to: ",column_value , type(error), error)
|
to: ",column_value , type(error), error)
|
||||||
@@ -651,7 +733,6 @@ def check_is_port_open(host, port):
|
|||||||
|
|
||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
|
|
||||||
def get_wordpress_post_id_and_link(postname,headers2):
|
def get_wordpress_post_id_and_link(postname,headers2):
|
||||||
response = requests.get(env.wpAPI+"/posts?search="+postname, headers=headers2,timeout=40)
|
response = requests.get(env.wpAPI+"/posts?search="+postname, headers=headers2,timeout=40)
|
||||||
result = response.json()
|
result = response.json()
|
||||||
@@ -726,17 +807,24 @@ def get_wordpress_featured_photo_id(post_id):
|
|||||||
##################################################################################################
|
##################################################################################################
|
||||||
|
|
||||||
def post_to_x2(title, content, date, rating, address, picslist, instasession):
|
def post_to_x2(title, content, date, rating, address, picslist, instasession):
|
||||||
"""Post to twitter
|
"""
|
||||||
|
Post to x2.
|
||||||
|
|
||||||
|
This function posts content to a social media platform using the provided data.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
title (_type_): _description_
|
title (str): The title of the post.
|
||||||
content (_type_): _description_
|
content (str): The content of the post.
|
||||||
date (_type_): _description_
|
date (str): The date of the post.
|
||||||
rating (_type_): _description_
|
rating (int): The rating of the post.
|
||||||
address (_type_): _description_
|
address (str): The address associated with the post.
|
||||||
picslist (_type_): _description_
|
picslist (list): A list of pictures for the post.
|
||||||
instasession (_type_): _description_
|
instasession: The Instagram session for posting.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The media ID of the posted content.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
pics = ((picslist[1:-1]).replace("'","")).split(",")
|
pics = ((picslist[1:-1]).replace("'","")).split(",")
|
||||||
# Replace the following strings with your own keys and secrets
|
# Replace the following strings with your own keys and secrets
|
||||||
CONSUMER_KEY = env.x_consumer_key
|
CONSUMER_KEY = env.x_consumer_key
|
||||||
@@ -1380,7 +1468,7 @@ def process_reviews(outputs):
|
|||||||
print (' Facebook: Skipping posting for ',processrow[1].value,' previously written')
|
print (' Facebook: Skipping posting for ',processrow[1].value,' previously written')
|
||||||
if env.xtwitter:
|
if env.xtwitter:
|
||||||
#if writtento["xtwitter"] == 0:
|
#if writtento["xtwitter"] == 0:
|
||||||
# if Posts.query.filter(Posts.name.xtwitter.op('!=')(1)).first()
|
# # if Posts.query.filter(Posts.name.xtwitter.op('!=')(1)).first()
|
||||||
if outputs['postssession'].query(Posts).filter(Posts.name == processrow[1].value,Posts.xtwitter != 1):
|
if outputs['postssession'].query(Posts).filter(Posts.name == processrow[1].value,Posts.xtwitter != 1):
|
||||||
if xtwittercount < env.postsperrun:
|
if xtwittercount < env.postsperrun:
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user