Coverage for src/crawler/by_source/ems_crawler.py: 89%

66 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1from urllib.parse import urljoin 

2 

3from bs4 import BeautifulSoup 

4from ptf.model_data import create_abstract, create_articledata, create_subj 

5 

6from crawler.abstract_crawlers.matching_crawler import MatchingCrawler 

7from crawler.utils import cleanup_str, regex_to_dict 

8 

9 

10class EmsCrawler(MatchingCrawler): 

11 source_name = "EMS Press" 

12 source_domain = "EMS" 

13 source_website = "https://ems.press/" 

14 

15 issue_re_1 = ( 

16 r"Vol\. (?P<volume>\d+),(?: No\. (?P<number>\d+),)?pp\. (?P<fpage>\d+)–(?P<lpage>\d+)" 

17 ) 

18 issue_re_2 = r"Vol\. (?P<volume>\d+),(?: No\. (?P<number_1>\d+)/(?P<number_2>\d+),)?pp\. (?P<fpage>\d+)–(?P<lpage>\d+)" 

19 

20 def parse_collection_content(self, content): 

21 xissues = [] 

22 soup = BeautifulSoup(content, "html.parser") 

23 issues = soup.select("a.issue-title") 

24 for issue in issues: 

25 previous_tag = issue.parent.find_previous_sibling("h2", {"class": "volume-title"}) 

26 volume_tag = previous_tag.select_one(".volume-year") 

27 volume_year = int(volume_tag.text) 

28 

29 try: 

30 issue_group = regex_to_dict( 

31 self.issue_re_1, issue.text, error_msg="Couldn't parse issue data" 

32 ) 

33 except ValueError: 

34 issue_group = regex_to_dict( 

35 self.issue_re_2, issue.text, error_msg="Couldn't parse issue data" 

36 ) 

37 

38 issue_href = issue.get("href") 

39 if not isinstance(issue_href, str): 39 ↛ 40line 39 didn't jump to line 40 because the condition on line 39 was never true

40 raise ValueError("Couldn't parse issue url") 

41 

42 xissues.append( 

43 self.create_xissue( 

44 urljoin(self.source_website, issue_href), 

45 volume_year, 

46 issue_group["volume"], 

47 # issue_number=issue_group.get("number_1")+ "-" + issue_group.get("number_2"), 

48 issue_group.get("number"), 

49 ) 

50 ) 

51 

52 return xissues 

53 

54 def parse_issue_content(self, content, xissue): 

55 soup = BeautifulSoup(content, "html.parser") 

56 articles = soup.select("article > a.unstyled") 

57 for index, article_tag in enumerate(articles): 

58 xarticle = create_articledata() 

59 xarticle.pid = "a" + str(index) 

60 article_href = article_tag.get("href") 

61 if not isinstance(article_href, str): 61 ↛ 62line 61 didn't jump to line 62 because the condition on line 61 was never true

62 raise ValueError("Couldn't parse article href") 

63 xarticle.url = urljoin(self.source_website, article_href) 

64 xissue.articles.append(xarticle) 

65 

66 def parse_article_content(self, content, xissue, xarticle, url): 

67 soup = BeautifulSoup(content, "html.parser") 

68 

69 self.get_metadata_using_citation_meta( 

70 xarticle, xissue, soup, ["author", "doi", "title", "pdf", "page", "title"] 

71 ) 

72 

73 # Abstract 

74 abstract_tag = soup.select_one("div.formatted-text > p") 

75 

76 if abstract_tag: 76 ↛ 81line 76 didn't jump to line 81 because the condition on line 76 was always true

77 abstract_text = cleanup_str(abstract_tag.text) 

78 xarticle.abstracts.append(create_abstract(lang="en", value_tex=abstract_text)) 

79 

80 # Keywords 

81 keywords_tag = soup.select("ul.keywords > li") 

82 for k_tag in keywords_tag: 

83 kwd_type = "" 

84 if k_tag.parent.previous_sibling.text == "Mathematics Subject Classification": 

85 kwd_type = "msc" 

86 kwd_text = cleanup_str(k_tag.text) 

87 if kwd_text != "": 87 ↛ 82line 87 didn't jump to line 82 because the condition on line 87 was always true

88 keyword = create_subj(value=kwd_text, type=kwd_type) 

89 xarticle.kwds.append(keyword) 

90 

91 # Contributors ORCID 

92 contributors_tag = soup.find("div", class_="person-group") 

93 if contributors_tag: 93 ↛ 102line 93 didn't jump to line 102 because the condition on line 93 was always true

94 contributor_list = contributors_tag.find_all("section", class_="person") 

95 if len(contributor_list) == len(xarticle.contributors): 95 ↛ 102line 95 didn't jump to line 102 because the condition on line 95 was always true

96 for i in range(len(contributor_list)): 

97 orcid_tag = contributor_list[i].find("a", string="ORCID") 

98 orcid_url = orcid_tag["href"] if orcid_tag else None 

99 if orcid_url: 

100 orcid = orcid_url.split("/")[-1] 

101 xarticle.contributors[i]["orcid"] = orcid 

102 return xarticle