-
Notifications
You must be signed in to change notification settings - Fork 18
Expand file tree
/
Copy pathscrape.py
More file actions
71 lines (53 loc) · 1.25 KB
/
Copy pathscrape.py
File metadata and controls
71 lines (53 loc) · 1.25 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
from sys import argv
import requests
from bs4 import BeautifulSoup
script, _url = argv
def scrape(url):
r = requests.get(url)
doc = r.text
soup = BeautifulSoup(doc)
wikitext = soup.find(id="wikitext")
approvedTags = ["em", "strong", "a", "ul", "ol", "li"]
scraped = []
for c in wikitext.children:
try:
tagClass = c['class']
except TypeError:
tagClass = False
except KeyError:
tagClass = False
except AttributeError:
tagClass = False
try:
childTag = c.name
except TypeError:
childTag = False
except KeyError:
childTag = False
except AttributeError:
childTag = False
if childTag and childTag not in approvedTags:
pass
elif childTag and childTag in approvedTags:
if tagClass:
if tagClass[0] == "twikilink":
scraped.append(c.string.lower())
elif tagClass[0] == "urllink":
scraped.append(c.contents[0])
elif childTag == "ul" or childTag == "ol":
for d in c.children:
scraped.append(d.contents[0])
else:
scraped.append(c.string)
else:
scraped.append(c)
for i in scraped[:]:
if i == '\n':
scraped.remove(i)
# print scraped
article = "".join(scraped)
article = article.split('\n')
article = '\n\n'.join(article)
# print article
return article
print scrape(_url)