Coverage for src/crawler/by_source/ami_crawler.py: 87%

70 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1from urllib.parse import urljoin 

2 

3from bs4 import BeautifulSoup, Tag 

4from ptf.model_data import IssueData, create_articledata, create_contributor 

5 

6from crawler.abstract_crawlers.matching_crawler import MatchingCrawler 

7from crawler.utils import ( 

8 add_pdf_link_to_xarticle, 

9 add_source_link_to_xarticle, 

10 cleanup_str, 

11 regex_to_dict, 

12) 

13 

14 

15class AmiCrawler(MatchingCrawler): 

16 source_name = "Annales Mathematica et Informaticae website" 

17 source_domain = "AMI" 

18 source_website = "https://ami.uni-eszterhazy.hu/" 

19 

20 issue_re = r"Vol. \d+ \((?P<year>\d+)\)" 

21 pages_re = r"Pages: (?P<fpage>\d+)–(?P<lpage>\d+)" 

22 

23 def parse_collection_content(self, content): 

24 xissues = [] 

25 soup = BeautifulSoup(content, "html.parser") 

26 issues = soup.select("#realtart select[name='vol'] option") 

27 for issue in issues: 

28 vol_number = issue.get("value") 

29 if not isinstance(vol_number, str) or not vol_number.isdigit(): 

30 continue 

31 issue_dict = regex_to_dict( 

32 self.issue_re, issue.text, error_msg="Couldn't parse volume year" 

33 ) 

34 xissues.append( 

35 self.create_xissue( 

36 self.collection_url + "?vol=" + vol_number, 

37 int(issue_dict["year"]), 

38 vol_number, 

39 None, 

40 ) 

41 ) 

42 return xissues 

43 

44 def parse_issue_content(self, content, xissue): 

45 soup = BeautifulSoup(content, "html.parser") 

46 articles = soup.select("#realtart p.cikk") 

47 for index, article_tag in enumerate(articles): 

48 xissue.articles.append(self.parse_ami_article(article_tag, xissue, index)) 

49 

50 def parse_ami_article(self, article_tag: Tag, xissue: IssueData, index: int): 

51 if not xissue.pid: 51 ↛ 52line 51 didn't jump to line 52 because the condition on line 51 was never true

52 raise ValueError("You must set xissue.pid before parsing an article") 

53 if not xissue.url: 53 ↛ 54line 53 didn't jump to line 54 because the condition on line 53 was never true

54 raise ValueError("You must set xissue.url before parsing an article") 

55 

56 xarticle = create_articledata() 

57 xarticle.lang = "en" 

58 xarticle.pid = xissue.pid + "_a" + str(index) 

59 

60 add_source_link_to_xarticle(xarticle, xissue.url, self.source_domain) 

61 

62 # Title 

63 title_tag = article_tag.select_one("a[href^='./uploads']") 

64 if not title_tag: 64 ↛ 65line 64 didn't jump to line 65 because the condition on line 64 was never true

65 raise ValueError("Couldn't parse article title") 

66 xarticle.title_tex = title_tag.text 

67 

68 # PDF 

69 pdf_url = title_tag.get("href") 

70 if not isinstance(pdf_url, str): 70 ↛ 71line 70 didn't jump to line 71 because the condition on line 70 was never true

71 raise ValueError("Couldn't parse article href") 

72 pdf_url = urljoin(self.source_website, pdf_url) 

73 add_pdf_link_to_xarticle(xarticle, pdf_url) 

74 

75 title_tag.decompose() 

76 # DOI 

77 doi_tag = article_tag.select_one("a[href^='https://doi.org']") 

78 if doi_tag: 

79 xarticle.doi = doi_tag.text 

80 doi_tag.decompose() 

81 

82 # Pages 

83 pages_tag = article_tag.select_one("span.oldal") 

84 if not pages_tag: 84 ↛ 85line 84 didn't jump to line 85 because the condition on line 84 was never true

85 raise ValueError("Couldn't find pages") 

86 pages_group = regex_to_dict( 

87 self.pages_re, pages_tag.text, error_msg="Couldn't parse pages" 

88 ) 

89 xarticle.fpage = pages_group["fpage"] 

90 xarticle.lpage = pages_group["lpage"] 

91 

92 # Authors 

93 authors = None 

94 for child in article_tag.children: 94 ↛ 101line 94 didn't jump to line 101 because the loop on line 94 didn't complete

95 if not isinstance(child, str): 

96 continue 

97 child = cleanup_str(child) 

98 if child.startswith("by"): 

99 authors = child.removeprefix("by ") 

100 break 

101 if not authors: 101 ↛ 102line 101 didn't jump to line 102 because the condition on line 101 was never true

102 raise ValueError("Couldn't find authors") 

103 

104 authors = authors.split(", ") 

105 for a in authors: 

106 xarticle.contributors.append(create_contributor(string_name=a, role="author")) 

107 

108 return xarticle