-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathBiohunter.py
More file actions
132 lines (118 loc) · 5.56 KB
/
Copy pathBiohunter.py
File metadata and controls
132 lines (118 loc) · 5.56 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
'''
Goals:
Provide easy interface to ANB search API (a. user-entered parameters; b. auto generate from authors list in SQL)
Automate parsing of search results and downloading of relavent biographies
Analyze biographies in NER and visualize links (see degrees of francis bacon project?)
(???)
Components:
[]input taker (user input prompt) > parameters:
[]autoinput taker (sql/csv)
[]webpage grabber (search page)
[]biographical link collector (allow manual user parsing of authors? resolve name sharers)
[]webpage grabber (biographies)
[]text cleaner
'''
import requests
from bs4 import BeautifulSoup
#ANB Specific
def ANBseek(parameters):
'''
search api translator for Oxford ANB with ND proxy info.
available search parameters: fulltext, name, gender, Y birth Start, Y birth End, birth state, birth country
'''
if type(parameters) != dict:
return 0
else:
if parameters["birthstate"] != "":
parameters["birthplace"] = parameters["birthstate"]
elif parameters["birthcountry"] != "":
parameters["birthplace"] = parameters["birthcountry"]
else:
parameters["birthplace"] = ""
if parameters["StartYear"] != "":
parameters["startday"] = "1"
parameters["startmonth"] = "1"
else:
parameters["startday"] = ""
parameters["startmonth"] = ""
if parameters["EndYear"] != "":
parameters["endday"] = "31"
parameters["endmonth"] = "12"
else:
parameters["endday"] = ""
parameters["endmonth"] = ""
url = "http://www.anb.org.proxy.library.nd.edu/articles/bin/search.cgi?func=advanced_search&fulltext={0}&idxa=-at&idxb=-bib&field-Name={1}&field-gender={2}&realms_top=Writing+and+Publishing&field-Realm=Writing+and+Publishing&date-m-birthS={3}&date-d-birthS={4}&date-y-birthS={5}&date-m-birthE={6}&date-d-birthE={7}&date-y-birthE={8}&date-m-deathS=&date-d-deathS=&date-y-deathS=&date-m-deathE=&date-d-deathE=&date-y-deathE=&field-BirthPlace={9}&bp_state={10}&bp_country={11}&field-Contrib=&meta-dc=500".format("+".join(parameters["fulltext"].split()), "+".join(parameters["name"].split()), parameters["gender"], parameters["startmonth"], parameters["startday"], parameters["StartYear"], parameters["endmonth"], parameters["endday"], parameters["EndYear"], "+".join(parameters["birthplace"].split()), "+".join(parameters["birthstate"].split()), "+".join(parameters["birthcountry"].split()))
#download search page and read its contents.
#using Requests module since Oxford requires cookies to be accepted
page = requests.get(url)
searchtext = page.text
yummyloc = searchtext.find("Search Results List") + 27
nomoreyummy = searchtext.find("<!--back to the top-->")
washedtext = searchtext[yummyloc:nomoreyummy]
#TODO: count the number of results to determine whether there's more than one page of data
soup = BeautifulSoup(washedtext)
goodstuff = soup.ol.find_all("a")
link_list = []
for href in goodstuff:
link_list.append(href['href'])
count = 0 #trying to be cheap and write over link_list variable onto itself
for x in link_list:
if x.find("?") != -1:
stoppoint = x.find("?")-5
else:
stoppoint = len(x)-5
link_list[count] = "http://www.anb.org.proxy.library.nd.edu/articles/"+x[3:stoppoint]+"-print.html"
count += 1
return link_list
def ANBreap(victim):
writefile = ""
soupfile = requests.get(victim).text
yummyloc = soupfile.find("</TD>")+11
nomoreyummy = soupfile.find("<HR>")-15
soupfile = soupfile[yummyloc:nomoreyummy]
soup = BeautifulSoup(soupfile)
soupfile = soup.find_all("p")
for x in soupfile:
writefile += str(x)+"\n"
return writefile
def ANBredeemer(soul):
soup = BeautifulSoup(soul.replace("\n", " "))
repeat = len(soup.find_all("p"))
count = 0
while (count < repeat):
soup.p.append("\n")
soup.p.unwrap()
count += 1
return soup.get_text()
#End ANB Specific
def seeker(site):
'''collects search results from specified site and returns a list of links, then prints basic overview info to screen
'''
#test
parameters = {"name":"", "StartYear":"1790", "EndYear":"1900", "gender":"m", "birthstate":"Alabama", "birthcountry":"", "fulltext":""}
#this leaves room for other sites to be added later
if site == "ANB":
return ANBseek(parameters)
#not a problem now since user's being prompted to enter parameters, but might come in handy when we automate input to check for correct type of input
else:
print "input must be in the form of a dictionary"
return 0
def reaper(victims, site):
if site == "ANB":
return ANBreap(victims)
def baptist(goat, site):
if site == "ANB":
return ANBredeemer(goat)
#--------------------------execute--------------------------
if __name__ == "__main__":
bookoflife = seeker("ANB")
if bookoflife == 0:
print "exited with error"
else: #if the program has made it this far, the data should have been massaged into an appropriate format already, so no longer checking for errors (famous last words?)
repeat = len(bookoflife)
count = 0
while (count < repeat):
gold = baptist(reaper(bookoflife[count], "ANB"), "ANB") #got lazy
f = open("bio{0}.txt".format(str(count + 1)), 'w')
f.write(gold)
count += 1