Repository navigation
Expand file tree
/
Copy pathcheck.py
More file actions
297 lines (266 loc) · 13.7 KB
/
Copy pathcheck.py
File metadata and controls
297 lines (266 loc) · 13.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
import bibtexparser
import requests
from pyalex import Works, config as pyalexconfig
from bibtexparser.middlewares.fieldkeys import NormalizeFieldKeys
####################################
# Input: adapt the file name here: #
####################################
file_name = "bibfiles/examples.bib"
# pyalax settings, see documentation: https://github.com/J535D165/pyalex
pyalexconfig.email = "mail@example.com"
class colors:
RED = '\033[31m'
ENDC = '\033[m'
GREEN = '\033[32m'
YELLOW = '\033[33m'
BLUE = '\033[34m'
def normalize(text):
text = text.lower()
punctations = [".", ",", ";", ":", "?", "!", "{", "}"]
for replacing in punctations:
text = text.replace(replacing, "")
#fix spacing problems, e.g. trailing spaces or double spaces
text = " ".join(text.split())
return text
def doi_lookup(doi):
response = requests.get('https://doi.org/' + doi, headers={"Accept": "application/x-bibtex"})
if response.status_code != 200:
print("ERROR looking up DOI")
return
return response.content
# search in Google Books API
def search_gb(authors, title, year):
authors_string = ", ".join(authors)
search_query = year + " inauthor:" + authors_string + " intitle:" + title
param = {"q": search_query}
queryString = requests.compat.urlencode(param)
gb_base_url = "https://www.googleapis.com/books/v1/volumes?"
url = gb_base_url + queryString
response = requests.get(url)
if response.status_code != 200:
print("ERROR requesting Google Books API", url)
return
data = response.json()
totalItems = data["totalItems"]
print("Google Books: Found", totalItems, "matches for the combination title, year, authors -", url)
if totalItems > 0:
items = data["items"]
# check title
items_matching_title = [item for item in items if ("title" in item["volumeInfo"] and normalize(item["volumeInfo"]["title"]).lower() == normalize(title).lower())]
if len(items_matching_title) > 0:
print(colors.GREEN + " ✓" + colors.ENDC, len(items_matching_title), "match title exactly")
items = items_matching_title
else:
print(colors.RED + " ×" + colors.ENDC, "None of them match the title exactly", [item["volumeInfo"]["title"] for item in items if "title" in item["volumeInfo"]])
# check year
items_matching_year = [item for item in items if ("publishedDate" in item["volumeInfo"] and abs(int(item["volumeInfo"]["publishedDate"][:4]) - int(year)) <= 1)]
if len(items_matching_year) > 0:
print(colors.GREEN + " ✓" + colors.ENDC, len(items_matching_year), "match year +-1")
items = items_matching_year
else:
print(colors.RED + " ×" + colors.ENDC, "None of them match the year +-1", [item["volumeInfo"]["publishedDate"] for item in items])
# check authors
compare_authors(authors, items[0]["volumeInfo"]["authors"])
def search_openalex(title, publication_type, entry_year, entry_authors):
# some interpunction can be important, e.g. if 2.0 is part of the title
# but we need to replace "," as this gives an error otherwise
title = title.replace(",", "").replace("{", "").replace("}", "")
works = Works().search_filter(title=title)
total = works.count()
openalex_url = works.url
exact_title_matches = []
for result in works.get(per_page = 200):
if normalize(result["title"]).lower() == normalize(title).lower():
exact_title_matches.append(result)
print("OpenAlex: Found " + str(len(exact_title_matches)) + " exact title matches (overall " + str(total) + " search results) " + works.url)
results_openalex = exact_title_matches
if len(results_openalex) > 0:
# save source_types directly in seperated field for each result
for result in results_openalex:
result["source_types"] = []
for location in result["locations"]:
if location["source"]:
result["source_types"].append(location["source"]["type"])
# check type
results_openalex_sametype = [result for result in results_openalex if check_type(result["type"], result["source_types"], publication_type)]
if len(results_openalex_sametype) > 0:
print(colors.GREEN + " ✓" + colors.ENDC, len(results_openalex_sametype), "match also the type")
results_openalex = results_openalex_sametype
else:
print(colors.RED + " ×" + colors.ENDC, "None of them match also the type", [result["type"] for result in results_openalex])
# check year
if entry_year:
results_openalex_sameyear = [result for result in results_openalex if abs(int(entry_year.value) - int(result["publication_year"])) <= 1]
openalex_years = [result_openalex["publication_year"] for result_openalex in results_openalex]
if len(results_openalex_sameyear) > 0:
print(colors.GREEN + " ✓" + colors.ENDC, len(results_openalex_sameyear), "matches the same year +-1.", entry_year.value, "vs.", openalex_years)
results_openalex = results_openalex_sameyear
else:
print(colors.RED + " ×" + colors.ENDC, "years are different.", entry_year.value, "vs.", openalex_years)
# check authors
openalex_authors = [author["raw_author_name"] for author in results_openalex[0]["authorships"]]
if entry_authors:
bibtex_authors = entry_authors.value
compare_authors(bibtex_authors, openalex_authors)
print(" ", results_openalex[0]["id"])
def search_zdb(journal_name):
if "arXiv preprint" in journal_name:
print("arXiv preprint is not a journal article")
return
journal_name = journal_name.replace("/", "").replace("(", "").replace(")", "")
param = {"q": "tit=" + journal_name}
queryString = requests.compat.urlencode(param)
url = 'https://zeitschriftendatenbank.de/api/tit.jsonld?' + queryString
response = requests.get(url)
if response.status_code != 200:
print("ERROR requesting ZDB", url)
return
data = response.json()
if "statusCode" in data and data["statusCode"] != 200:
print("ERROR requesting ZDB: " + data["description"], url)
return
exact_matches = []
nResults = data["totalItems"]
for record in data["member"]:
if record:
# Often in the ZDB several variants of the title, short title are concatenated
# with a " : " as separator between, e.g. 'Electronic Journal of Graph Theory and Applications : EJGTA'
title_variants = record["title"].split(" : ")
title_variants.append(normalize(record["title"]))
for variant in title_variants:
if variant.lower() == normalize(journal_name).lower():
exact_matches.append(record["title"])
break
if len(exact_matches)>0:
zdb_status = colors.GREEN + "✓" + colors.ENDC
elif len(data["member"])>0:
zdb_status = colors.YELLOW + "?" + colors.ENDC
else:
zdb_status = colors.RED + "×" + colors.ENDC
print("ZDB:", zdb_status, "Found", str(len(exact_matches)), "exact matches (overall", str(len(data["member"])), "search results) for the journal name", journal_name)
def check_type(openalex_type, openalex_source_types, bibtex_type):
match bibtex_type:
case "book":
return openalex_type == "book"
case "inproceedings":
if "conference" in openalex_source_types:
return True
case "inbook" | "incollection":
return openalex_type == "book-chapter"
case "article":
return "journal" in openalex_source_types
case "misc":
freq_preprint = [1 for t in openalex_source_types if t == "preprint"]
if openalex_type in ["preprint", "dataset"] or len(freq_preprint) == len(openalex_source_types):
return True
return False
def compare_bibtex_values(key, original_value, value_to_compare):
original_value_saved = original_value
if key in ["title", "booktitle", "journal", "year", "volume", "issue", "pages"]:
if not value_to_compare:
print(colors.BLUE + " ☠ " + colors.ENDC, key, "missing", field.value)
else:
# normalize values depending on which key we are looking at
if key in ["title", "booktitle", "journal", "publisher"]:
original_value = normalize(original_value).lower()
value_to_compare = normalize(value_to_compare).lower()
elif key in ["year", "volume", "issue"]:
try:
original_value = int(original_value)
except:
original_value = -1
try:
value_to_compare = int(value_to_compare)
except:
print(colors.BLUE + " ☠ " + colors.ENDC, key, "not a number", value_to_compare)
value_to_compare = -1
elif key == "pages":
original_value = original_value.replace("\\xe2\\x80\\x93", "-").replace("--", "-").replace("–", "-")
value_to_compare = value_to_compare.replace("\\xe2\\x80\\x93", "-").replace("--", "-").replace("–", "-")
# comparison
if original_value == value_to_compare:
print(colors.GREEN + " ✓" + colors.ENDC, key, "matches:", original_value_saved)
else:
print(colors.RED + " ×" + colors.ENDC, key, "differs:", original_value, "=/=", value_to_compare)
def compare_authors(input_authors, comparison_authors):
# authors in inverse format lastname, firstname are reversed
for i, author in enumerate(comparison_authors):
if ", " in author:
reverse_author = " ".join(author.split(", ")[::-1])
comparison_authors[i] = reverse_author
for j, author in enumerate(input_authors):
if ", " in author:
reverse_author = " ".join(author.split(", ")[::-1])
input_authors[j] = reverse_author
# count number of authors from the input which also occur in the comparison
found_input_authors = 0
for author in input_authors:
if author in comparison_authors:
found_input_authors += 1
found_comparison_authors = 0
for author in comparison_authors:
if author in input_authors:
found_comparison_authors += 1
firstCheck = colors.GREEN + "✓" +colors.ENDC if found_input_authors == len(input_authors) else colors.RED + "×" + colors.ENDC
secondCheck = colors.GREEN + "✓" +colors.ENDC if found_comparison_authors == len(comparison_authors) else colors.RED + "×" + colors.ENDC
print(" " + firstCheck + secondCheck + " Found", found_input_authors, "of the", len(input_authors), "input authors resp.", found_comparison_authors, "of the", len(comparison_authors), "authors from the comparison.")
print(" ", input_authors)
print(" ", comparison_authors)
# besides the standard configuration of the bibtexparser (resolving strings + removing enclosing),
# we want also to following two middlewares and normalize the keys.
layers = [
bibtexparser.middlewares.LatexDecodingMiddleware(),
bibtexparser.middlewares.SeparateCoAuthors()
]
library = bibtexparser.parse_file(file_name, append_middleware=layers)
library = NormalizeFieldKeys().transform(library)
for entry in library.entries:
print("\n" + entry.key + " [" + entry.entry_type + "]" + "\n==============================")
if not "year" in entry:
if "date" in entry:
entry["year"] = entry["date"][:4]
else:
print(colors.BLUE + " ☠ " + colors.ENDC, "year is missing")
# DOI lookup
if "doi" in entry:
print("DOI found:", "https://doi.org/" + entry["doi"])
bibtex_from_doi = doi_lookup(entry["doi"])
if bibtex_from_doi:
data_from_doi = bibtexparser.parse_string(bibtex_from_doi)
entry_from_doi = data_from_doi.entries[0]
for field in entry_from_doi.fields:
key = field.key.replace("\\n", "")
key = key.strip()
value_to_compare = entry.get(key).value if entry.get(key) else None
compare_bibtex_values(key, field.value, value_to_compare)
# search exact title in OpenAlex
if "title" in entry:
results_openalex = search_openalex(entry["title"], entry.entry_type, entry.get("year"), entry.get("author"))
else:
print(colors.BLUE + "☠" + colors.ENDC + " title is missing")
# search journal in ZDB
if "journal" in entry:
journal_name = entry["journal"]
elif "journaltitle" in entry:
journal_name = entry["journaltitle"]
elif "shortjournal" in entry:
journal_name = entry["shorttitle"]
else:
journal_name = None
if entry.entry_type == "article":
if journal_name:
search_zdb(journal_name)
else:
print(colors.BLUE + "☠" + colors.ENDC + " journal name is missing")
# search book in GoogleBooks
if entry.entry_type == "book":
if "title" in entry and "year" in entry and "author" in entry:
gb_results = search_gb(entry["author"], entry["title"], entry["year"])
# url to manually search title + year + author in Google Scholar
if "author" in entry and "title" in entry and "year" in entry:
gs_base_url = "https://scholar.google.com/scholar?"
search_query = ". ".join([", ".join(entry["author"]), entry["title"], str(entry["year"])])
param = {"q": search_query}
queryString = requests.compat.urlencode(param)
print("GoogleScholar search manually:", gs_base_url + queryString)
#break