Coverage for src/crawler/by_source/ems_crawler.py: 89%
66 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1from urllib.parse import urljoin
3from bs4 import BeautifulSoup
4from ptf.model_data import create_abstract, create_articledata, create_subj
6from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
7from crawler.utils import cleanup_str, regex_to_dict
10class EmsCrawler(MatchingCrawler):
11 source_name = "EMS Press"
12 source_domain = "EMS"
13 source_website = "https://ems.press/"
15 issue_re_1 = (
16 r"Vol\. (?P<volume>\d+),(?: No\. (?P<number>\d+),)?pp\. (?P<fpage>\d+)–(?P<lpage>\d+)"
17 )
18 issue_re_2 = r"Vol\. (?P<volume>\d+),(?: No\. (?P<number_1>\d+)/(?P<number_2>\d+),)?pp\. (?P<fpage>\d+)–(?P<lpage>\d+)"
20 def parse_collection_content(self, content):
21 xissues = []
22 soup = BeautifulSoup(content, "html.parser")
23 issues = soup.select("a.issue-title")
24 for issue in issues:
25 previous_tag = issue.parent.find_previous_sibling("h2", {"class": "volume-title"})
26 volume_tag = previous_tag.select_one(".volume-year")
27 volume_year = int(volume_tag.text)
29 try:
30 issue_group = regex_to_dict(
31 self.issue_re_1, issue.text, error_msg="Couldn't parse issue data"
32 )
33 except ValueError:
34 issue_group = regex_to_dict(
35 self.issue_re_2, issue.text, error_msg="Couldn't parse issue data"
36 )
38 issue_href = issue.get("href")
39 if not isinstance(issue_href, str): 39 ↛ 40line 39 didn't jump to line 40 because the condition on line 39 was never true
40 raise ValueError("Couldn't parse issue url")
42 xissues.append(
43 self.create_xissue(
44 urljoin(self.source_website, issue_href),
45 volume_year,
46 issue_group["volume"],
47 # issue_number=issue_group.get("number_1")+ "-" + issue_group.get("number_2"),
48 issue_group.get("number"),
49 )
50 )
52 return xissues
54 def parse_issue_content(self, content, xissue):
55 soup = BeautifulSoup(content, "html.parser")
56 articles = soup.select("article > a.unstyled")
57 for index, article_tag in enumerate(articles):
58 xarticle = create_articledata()
59 xarticle.pid = "a" + str(index)
60 article_href = article_tag.get("href")
61 if not isinstance(article_href, str): 61 ↛ 62line 61 didn't jump to line 62 because the condition on line 61 was never true
62 raise ValueError("Couldn't parse article href")
63 xarticle.url = urljoin(self.source_website, article_href)
64 xissue.articles.append(xarticle)
66 def parse_article_content(self, content, xissue, xarticle, url):
67 soup = BeautifulSoup(content, "html.parser")
69 self.get_metadata_using_citation_meta(
70 xarticle, xissue, soup, ["author", "doi", "title", "pdf", "page", "title"]
71 )
73 # Abstract
74 abstract_tag = soup.select_one("div.formatted-text > p")
76 if abstract_tag: 76 ↛ 81line 76 didn't jump to line 81 because the condition on line 76 was always true
77 abstract_text = cleanup_str(abstract_tag.text)
78 xarticle.abstracts.append(create_abstract(lang="en", value_tex=abstract_text))
80 # Keywords
81 keywords_tag = soup.select("ul.keywords > li")
82 for k_tag in keywords_tag:
83 kwd_type = ""
84 if k_tag.parent.previous_sibling.text == "Mathematics Subject Classification":
85 kwd_type = "msc"
86 kwd_text = cleanup_str(k_tag.text)
87 if kwd_text != "": 87 ↛ 82line 87 didn't jump to line 82 because the condition on line 87 was always true
88 keyword = create_subj(value=kwd_text, type=kwd_type)
89 xarticle.kwds.append(keyword)
91 # Contributors ORCID
92 contributors_tag = soup.find("div", class_="person-group")
93 if contributors_tag: 93 ↛ 102line 93 didn't jump to line 102 because the condition on line 93 was always true
94 contributor_list = contributors_tag.find_all("section", class_="person")
95 if len(contributor_list) == len(xarticle.contributors): 95 ↛ 102line 95 didn't jump to line 102 because the condition on line 95 was always true
96 for i in range(len(contributor_list)):
97 orcid_tag = contributor_list[i].find("a", string="ORCID")
98 orcid_url = orcid_tag["href"] if orcid_tag else None
99 if orcid_url:
100 orcid = orcid_url.split("/")[-1]
101 xarticle.contributors[i]["orcid"] = orcid
102 return xarticle