Coverage for src/crawler/by_source/cup_crawler.py: 10%
200 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import logging
2import re
3from urllib.parse import urljoin
5from bs4 import BeautifulSoup, Tag
6from ptf.cmds.xml.xml_utils import escape
7from ptf.model_data import create_abstract, create_articledata, create_contributor
9from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
10from crawler.cmds.mixed_citation import (
11 ExtLinkXml,
12 GenericRefElement,
13 MixedCitation,
14)
15from crawler.utils import cleanup_str, regex_to_dict
17logger = logging.getLogger(__name__)
20class CupCrawler(MatchingCrawler):
21 source_name = "Cambridge University Press"
22 source_domain = "CUP"
23 source_website = "https://www.cambridge.org/core/"
25 issue_re = r"Issue (?P<issue>\S+)"
26 issue_error_re = r"Volume (?P<issue_nb>\d+)"
27 volume_re = r"Volume (?P<volume>\d+)"
28 archive_volume_re = r"Vol (?P<volume>\d+)"
29 archive_year_re = r"Archive content \n\n\n (?P<year>\S+)"
31 pid_year_restrictions = {
32 "GLMJ": 6,
33 "CJM": 6,
34 "CMB": 6,
35 }
37 def parse_collection_content(self, content):
38 xissues = []
39 soup = BeautifulSoup(content, "html.parser")
41 volumes_tag = soup.select(
42 "div.journal-all-issues > ul > li > div.content > ul.accordion > li.accordion-navigation"
43 )
44 for volume_tag in volumes_tag:
45 issue_defaut_nb = "1"
46 volume = volume_tag.select_one("a")
47 if volume is None:
48 raise ValueError("Couldn't parse volume tag")
50 try:
51 volume_group = regex_to_dict(
52 self.volume_re, volume.text, error_msg="Couldn't parse volume number"
53 )
54 except ValueError:
55 try:
56 volume_group = regex_to_dict(
57 self.archive_volume_re,
58 volume.text,
59 error_msg="Couldn't parse volume number",
60 )
61 except ValueError:
62 raise ValueError(f"Couldn't parse volume number from text: {volume.text}")
64 issues_tag = volume_tag.select("div > ul > li > ul > li > a")
66 ## If no issue listed : we consider the volume has only one issue
67 if not issues_tag:
68 issue_href = volume.get("href")
69 year_span = volume.select_one("span.date")
70 if not year_span:
71 raise ValueError("Couldn't parse year for volume with no issue")
72 year = year_span.text.split(" ")[-1]
73 xissues.append(
74 self.create_xissue(
75 urljoin(self.source_website, issue_href),
76 int(year),
77 volume_group.get("volume"),
78 "1",
79 )
80 )
81 continue
83 # Get all the volume listed issues
84 for issue_tag in issues_tag:
85 issue_nb, issue_href, issue_year, issue_defaut_nb = self.get_issue_data(
86 issue_tag, issue_defaut_nb
87 )
88 # # Cambridge has declared articles younger than 5 not as open access
89 # if issue_year < current_year:
90 xissues.append(
91 self.create_xissue(
92 urljoin(self.source_website, issue_href),
93 issue_year,
94 volume_group.get("volume"),
95 issue_nb,
96 )
97 )
98 return xissues
100 def get_issue_data(self, issue_tag, default_issue_nb):
101 """
102 Get issue number in classic case but also in the special case of volume 27 with no issue number (defaults to issue 1)
103 """
104 year_span = issue_tag.select_one("span.date")
105 if not year_span:
106 raise ValueError("Couldn't parse year for issue")
107 year = int(year_span.text.split(" ")[-1])
109 issue_href = issue_tag.get("href")
110 if not isinstance(issue_href, str):
111 raise ValueError("Couldn't parse issue href")
113 try:
114 issue = regex_to_dict(
115 self.issue_re, issue_tag.text, error_msg="Couldn't parse issue number"
116 )
117 except ValueError:
118 try:
119 issue = regex_to_dict(
120 self.issue_error_re, issue_tag.text, error_msg="Couldn't parse issue number"
121 )
122 except ValueError:
123 raise ValueError(f"Couldn't parse issue number from text: {issue_tag.text}")
125 issue_nb = issue.get("issue")
126 return issue_nb, issue_href, year, default_issue_nb
128 def parse_issue_content(self, content, xissue):
129 soup = BeautifulSoup(content, "html.parser")
130 articles = soup.select("div.representation")
131 article_number = 0
133 for article in articles: # if one or more article not in open access, issue not fetched
134 if "Get access" in article.text:
135 print(f"deleete xissue {xissue.fyear}")
136 return
138 for article in articles:
139 if (
140 article.select_one(".access-modal > .status.open-access > .icon.open-access")
141 is None # open access icon
142 and article.select_one(".access-modal > .status.entitled > .icon.access-icon")
143 is None # access without licence CCO
144 ):
145 logger.debug("Article is not accessible. skipping.")
146 continue
147 item = article.select_one(".access-modal")
148 if "Cover and " in item.parent.parent.parent.parent.get_text():
149 logger.debug("This is cover and front, not article. skipping.")
150 continue
151 if "Index to Volume" in item.parent.parent.parent.parent.get_text():
152 logger.debug("This is Index, not article. skipping.")
153 continue
154 xarticle = create_articledata()
155 article_href = article.select_one("a.part-link").get("href")
156 if not isinstance(article_href, str):
157 raise ValueError("Couldn't parse article href")
158 xarticle.url = urljoin(self.source_website, article_href)
159 xarticle.pid = "a" + str(article_number)
160 xissue.articles.append(xarticle)
161 article_number += 1
163 has_pagination = soup.select_one("ul.pagination a:-soup-contains-own('Next »')")
164 if has_pagination:
165 pagination_link = has_pagination.get("href")
166 if isinstance(pagination_link, str):
167 page_url = urljoin(xissue.url, pagination_link)
168 content = self.download_file(page_url)
170 self.parse_issue_content(content, xissue)
172 def parse_article_content(self, content, xissue, xarticle, url):
173 soup = BeautifulSoup(content, "html.parser")
175 self.get_metadata_using_citation_meta(
176 xarticle,
177 xissue,
178 soup,
179 [
180 "pdf",
181 "page",
182 "doi",
183 "publisher",
184 "keywords",
185 "references",
186 ],
187 )
189 ## Title
190 title_tag = soup.select_one("hgroup > h1")
191 if title_tag is None:
192 raise ValueError(f"Couldn't parse article title for article with url: {xarticle.url}")
193 xarticle.title_tex = cleanup_str(title_tag.text)
195 ## Abstract
196 abstract_tag = soup.select_one("div.abstract")
198 if abstract_tag:
199 abstract = cleanup_str(abstract_tag.text)
200 xarticle.abstracts.append(create_abstract(value_tex=abstract, lang=xarticle.lang))
201 else:
202 logger.info(f"No abstract found for article with url: {xarticle.url}")
204 ## keywords
205 keywords_tag = soup.select_one("div.keywords")
206 keywords = keywords_tag.select("span") if keywords_tag else []
207 for keyword in keywords:
208 xarticle.kwds.append(
209 {"type": "", "lang": xarticle.lang, "value": cleanup_str(keyword.text)}
210 )
212 ## Contributors name doi email
213 self.parse_cup_contributors(soup, xarticle)
215 references_list = soup.select_one("#references-list")
216 if references_list:
217 xarticle.bibitems = self.parse_cambridge_references(references_list)
218 return xarticle
220 def parse_cup_contributors(self, soup, xarticle):
221 # Fetch ORCIDs [Name, ORCID]
222 contributors = soup.select_one("div.contributors-details")
223 if not contributors:
224 raise ValueError("Couldn't parse contributors")
226 orcid_by_name = {}
227 for orcid_link in contributors.find_all("a", {"data-test-orcid": True}):
228 name = orcid_link["data-test-orcid"]
229 href = orcid_link.get("href", "")
230 orcid_id = href.rstrip("/").split("/")[-1] if href else None
231 orcid_by_name[name] = orcid_id
233 # Fetch Emails [Name, Email]
234 email_by_name = {}
235 for corresp in contributors.find_all(class_="corresp"):
236 mailto = corresp.find("a", href=re.compile(r"^mailto:"))
237 if mailto:
238 email = mailto["href"].replace("mailto:", "")
239 # Le nom du correspondant est souvent juste avant dans le texte
240 # On cherche dans les blocs .author le lien corresp
241 email_by_name["__corresp__"] = email # sera affiné ci-dessous
243 # Fetch Authors
244 for author_block in contributors.find_all(attrs={"data-test-author": True}):
245 string_name = author_block["data-test-author"]
247 # Split name into first and last name
248 parts = string_name.strip().split()
249 if len(parts) >= 2:
250 first_name = " ".join(parts[:-1])
251 last_name = parts[-1]
252 else:
253 first_name = ""
254 last_name = string_name
256 # ORCID
257 orcid = orcid_by_name.get(string_name)
259 # Email
260 email = ""
261 mailto_tag = author_block.find("a", href=re.compile(r"^mailto:"))
262 if mailto_tag:
263 email = mailto_tag["href"].replace("mailto:", "")
265 xarticle.contributors.append(
266 create_contributor(
267 role="author",
268 string_name=string_name,
269 first_name=first_name,
270 last_name=last_name,
271 orcid=orcid,
272 email=email,
273 )
274 )
275 return xarticle
277 def parse_cambridge_references(self, soup: Tag):
278 bibitems = []
279 for item in soup.select(".circle-list__item"):
280 citation_builder = MixedCitation()
281 label_tag = item.select_one(".circle-list__item__number")
282 if label_tag:
283 citation_builder.label = escape(cleanup_str(label_tag.text))
284 citation_content = item.select_one(".circle-list__item__grouped__content")
285 if citation_content:
286 self.parse_cambridge_ref_nodes(citation_content, citation_builder)
288 # Group all StringNames into one PersonGroup object
289 persongroup_builder = GenericRefElement()
290 persongroup_builder.name = "person-group"
291 # Index of StringNames objects
292 i = [
293 index
294 for index, element in enumerate(citation_builder.elements)
295 if isinstance(element, GenericRefElement) and element.name == "string-name"
296 ]
297 if len(i) > 0:
298 persongroup_builder.elements = citation_builder.elements[i[0] : i[-1] + 1]
299 del citation_builder.elements[i[0] : i[-1] + 1]
300 citation_builder.elements.insert(i[0], persongroup_builder)
302 bibitems.append(citation_builder.get_jats_ref())
303 return bibitems
305 def parse_cambridge_ref_nodes(
306 self,
307 current_tag: Tag,
308 current_builder: GenericRefElement,
309 ):
310 "recursive function that parses references tags"
311 for element in current_tag.children:
312 if isinstance(element, str):
313 current_builder.elements.append(escape(element))
314 continue
315 if isinstance(element, Tag):
316 tag_class = element.get("class")
317 if isinstance(tag_class, list):
318 if len(tag_class) > 0:
319 tag_class = tag_class[0]
320 else:
321 tag_class = None
323 if not tag_class:
324 continue
325 if tag_class in ("mathjax-tex-wrapper", "aop-lazy-load-image"):
326 continue
327 if element.name == "a":
328 href = element.get("href")
329 if isinstance(href, str):
330 current_builder.elements.append(" ")
331 current_builder.elements.append(
332 ExtLinkXml(escape(href), escape(element.text))
333 )
334 continue
336 if tag_class in [
337 "surname",
338 "given-names",
339 "string-name",
340 "person-group",
341 "publisher-name",
342 "source",
343 "volume",
344 "year",
345 "fpage",
346 "lpage",
347 "article-title",
348 "issue",
349 "chapter-title",
350 "inline-formula",
351 "collab",
352 "alternatives",
353 "italic",
354 "publisher-loc",
355 "roman",
356 "edition",
357 "suffix",
358 ]:
359 refnode_builder = GenericRefElement()
360 refnode_builder.name = tag_class
361 current_builder.elements.append(refnode_builder)
362 self.parse_cambridge_ref_nodes(element, refnode_builder)
363 continue
365 self.logger.warning(f"Couldn't insert tag into mixed citation : {tag_class}")
366 current_builder.elements.append(escape(element.text))