-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscrape_text.py
More file actions
43 lines (40 loc) · 993 Bytes
/
Copy pathscrape_text.py
File metadata and controls
43 lines (40 loc) · 993 Bytes
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
import urllib
from bs4 import BeautifulSoup
import re
import random
#bank
words = []
joined = " "
#extractor, display text
url = raw_input("enter full url for extraction: ")
html = urllib.urlopen(url).read()
soup = BeautifulSoup(html, "html.parser")
[x.extract() for x in soup.find_all('script')]
text = soup.get_text(" ", strip=True)
print text
print " "
loop = 1
while loop == 1:
#inputter
input = raw_input("analyze: ")
#analyzer
if input != "":
regex = r'[^.?!]*(?<=[.?\s!])'+input+'(?=[\s.?!])[^.?!]*[.?!]'
result = re.findall(regex, text)
print result
if input == "":
regex = r'([A-Z][^\.!?]*[\.!?])'
result = re.findall(regex, text)
sentence = random.choice(result)
print sentence
#adder
final_input = raw_input("select: ")
words.append(final_input)
#shower
print words
#exiter
if input == " ":
print " "
print " ".join(words)
print " "
loop = 0