-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathparse.py
More file actions
115 lines (98 loc) · 3.83 KB
/
Copy pathparse.py
File metadata and controls
115 lines (98 loc) · 3.83 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
'''
Takes a plaintext file of unparsed bibliography entries and parses them using
AnyStyle.io. Then performs a search for the title on the HathiTrust and returns
matching IDs. The final data is output as a JSON struct for use on the web,
particularly with the corpus builder tool at
http://github.com/inpho/corpus-builder
Run with 'python darparse.py -h' to see a list of command arguments.
'''
import json
from time import sleep
from urllib2 import urlopen
from urllib import quote_plus
import os.path
import xmlrpclib
import rython
from codecs import open
ctx = rython.RubyContext(requires=["rubygems", "anystyle/parser"])
ctx("Encoding.default_internal = 'UTF-8'")
ctx("Encoding.default_external = 'UTF-8'")
anystyle = ctx("Anystyle.parser")
def parse_citations_from_file(citation_file):
return_list = []
with open(citation_file, 'r') as readfile:
for i, line in enumerate(readfile):
line = line.decode('latin-1').encode("utf-8").strip()
if line:
try:
parsed = anystyle.parse(line)[0]
return_list.append({'original' : line, 'parsed' : parsed})
except xmlrpclib.Fault:
return_list.append({'original' : line, 'parsed' : None})
print "error on line %d: %s" % (i, line)
return return_list
def search(title, sleep_time=1):
""" Queries the HTRC Solr index with a title and returns the resulting metadata.
Documentation: http://www.hathitrust.org/htrc/solr-api
"""
# TODO: Parameterize hostname
solr ="http://chinkapin.pti.indiana.edu:9994/solr/meta/select/?q=title:%s" % quote_plus(title)
solr += "&wt=json" ## retrieve JSON results
# TODO: exception handling
try:
data = json.load(urlopen(solr))
except ValueError :
print "No result found for " + title
return
if sleep:
sleep(sleep_time) ## JUST TO MAKE SURE WE ARE THROTTLED
try:
return data['response']['docs'][0]
except IndexError:
return None
def populate_htrc(citations):
for citation in citations:
citation['htrc_id'] = None
citation['htrc_md'] = None
if citation['parsed']:
title = citation['parsed'].get('title')
if title:
title = title.replace("/", "")
title = title.replace(":", "")
title = title.replace("[", "")
title = title.replace("]", "")
htrc_md = search(title.encode('utf-8'))
if htrc_md:
citation['htrc_md'] = htrc_md
citation['htrc_id'] = htrc_md.get('id')
return citations
def extant_file(x):
"""
'Type' for argparse - checks that file exists but does not open.
"""
if not os.path.isfile(x):
raise argparse.ArgumentError("%s does not exist" % x)
return x
if __name__ == '__main__':
from argparse import ArgumentParser
parser = ArgumentParser()
parser.add_argument('-o', dest='output', help="output file", default=None)
parser.add_argument('citation_file', help="citation file",
type=extant_file)
args = parser.parse_args()
print "Parsing Citations..."
citations = parse_citations_from_file(args.citation_file)
print "Retrieving HTRC IDs..."
citations = populate_htrc(citations)
print "Printing output file..."
while not args.output:
args.output = raw_input("Output filename: www/").strip()
if not args.output.startswith('www/'):
args.output = 'www/' + args.output
with open(args.output, 'wb') as output_file:
json.dump(citations, output_file)
print "TIP: launch Corpus Builder with:"
print "python server.py -p 9024"
print "\n"
print "TIP: Navigate your browser to:"
print "http://localhost:9024/?corpus="+args.output.replace('www/','')