forked from rhnvrm/ConstAssemblyBot
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathbot.py
More file actions
174 lines (144 loc) · 5.21 KB
/
Copy pathbot.py
File metadata and controls
174 lines (144 loc) · 5.21 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
import os
import tweepy
import logging
import re
# Config
MAX_CHAR_FOR_TWEET = 240
BOTNAME = "@ConstAssembly"
logging.basicConfig(filename="bot.log", level=logging.DEBUG)
def create_api():
consumer_key = os.getenv("CONSUMER_KEY")
consumer_secret = os.getenv("CONSUMER_SECRET")
access_token = os.getenv("ACCESS_TOKEN")
access_token_secret = os.getenv("ACCESS_SECRET")
auth = tweepy.OAuthHandler(consumer_key, consumer_secret)
auth.set_access_token(access_token, access_token_secret)
api = tweepy.API(auth, wait_on_rate_limit=True, wait_on_rate_limit_notify=True)
try:
api.verify_credentials()
except Exception as e:
logging.error("Error creating API", exc_info=True)
raise e
logging.info("API created")
return api
# -*- coding: utf-8 -*-
alphabets = "([A-Za-z])"
prefixes = "(Mr|St|Mrs|Ms|Dr|No)[.]"
suffixes = "(Inc|Ltd|Jr|Sr|Co)"
starters = "(Mr|Mrs|Ms|Dr|He\s|She\s|It\s|They\s|Their\s|Our\s|We\s|But\s|However\s|That\s|This\s|Wherever)"
acronyms = "([A-Z][.][A-Z][.](?:[A-Z][.])?)"
websites = "[.](com|net|org|io|gov)"
def split_into_sentences(text):
text = " " + text + " "
text = text.replace("\n", " ")
text = re.sub(prefixes, "\\1<prd>", text)
text = re.sub(websites, "<prd>\\1", text)
if "Ph.D" in text:
text = text.replace("Ph.D.", "Ph<prd>D<prd>")
text = re.sub("\s" + alphabets + "[.] ", " \\1<prd> ", text)
text = re.sub(acronyms + " " + starters, "\\1<stop> \\2", text)
text = re.sub(
alphabets + "[.]" + alphabets + "[.]" + alphabets + "[.]",
"\\1<prd>\\2<prd>\\3<prd>",
text,
)
text = re.sub(r'\.+', ".", text) # replace multiple dots with a single dot
text = re.sub(alphabets + "[.]" + alphabets + "[.]", "\\1<prd>\\2<prd>", text)
text = re.sub(" " + suffixes + "[.] " + starters, " \\1<stop> \\2", text)
text = re.sub(" " + suffixes + "[.]", " \\1<prd>", text)
text = re.sub(" " + alphabets + "[.]", " \\1<prd>", text)
if "”" in text:
text = text.replace(".”", "”.")
if '"' in text:
text = text.replace('."', '".')
if "!" in text:
text = text.replace('!"', '"!')
if "?" in text:
text = text.replace('?"', '"?')
text = text.replace(".", ".<stop>")
text = text.replace("?", "?<stop>")
text = text.replace("!", "!<stop>")
text = text.replace("<prd>", ".")
sentences = text.split("<stop>")
sentences = sentences[:-1]
sentences = [s.strip() for s in sentences]
return sentences
def split_text(tweet_text):
# Split words at spaces, ignore newlines
words = tweet_text.split(" ")
tweet = words[0] # Assign First word
for word in words[1:]:
if len(tweet) + len(word) + 1 > MAX_CHAR_FOR_TWEET:
# current line + word + 1 space exceeds single tweet
# return current line
yield tweet.strip()
tweet = word # Start ext tweet
else:
# There's more room left, add next word!
tweet = tweet + ' ' + word
yield tweet.strip()
def run():
api = create_api()
# read all lines
f_data = open("data.txt")
lines = f_data.readlines()
f_data.close()
# read last line
f = open("last_line.txt")
last_line_data = f.readlines()
f.close()
# get line to tweet
line_to_tweet = int(last_line_data[0])
line_char = int(last_line_data[1])
pending_tweet = ""
# check for ignorable lines
while True:
line = lines[line_to_tweet]
if line == "\n":
line_to_tweet += 1
else:
break
# if it matches with 1.1.1 format, skip line, add name, save to file
if re.match("^\d*\.\d*\.\d*", line):
line_to_tweet += 1
pending_tweet = lines[line_to_tweet]
line_to_tweet += 1
of = open("last_line.txt", "w")
of.writelines([str(line_to_tweet), "\n", str(line_char)])
of.close()
# check if it is a sentence
sentences = split_into_sentences(lines[line_to_tweet][line_char:])
if len(sentences) > 0:
first_sentence = sentences[0]
if len(first_sentence) != len(lines[line_to_tweet]):
line_char = lines[line_to_tweet].find(first_sentence) + len(first_sentence)
pending_tweet += first_sentence
if len(sentences) == 1:
line_char = 0
line_to_tweet += 1
else:
pending_tweet += lines[line_to_tweet]
line_to_tweet += 1
line_char = 0
# create tweet
logging.info("sending tweet: ===\n" + pending_tweet + "\n===")
prev_status = None
for out_tweet in split_text(pending_tweet):
try:
if prev_status == None:
prev_status = api.update_status(status=out_tweet)
else:
prev_status = api.update_status(
status=out_tweet, in_reply_to_status_id=prev_status.id
)
except tweepy.TweepError as error:
if error.api_code == 187:
# Do something special
api.update_status(status=out_tweet + ".")
else:
raise error
of = open("last_line.txt", "w")
of.writelines([str(line_to_tweet), "\n", str(line_char)])
of.close()
if __name__ == '__main__':
run()