Coverage for src/crawler/by_source/ami_crawler.py: 87%
70 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1from urllib.parse import urljoin
3from bs4 import BeautifulSoup, Tag
4from ptf.model_data import IssueData, create_articledata, create_contributor
6from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
7from crawler.utils import (
8 add_pdf_link_to_xarticle,
9 add_source_link_to_xarticle,
10 cleanup_str,
11 regex_to_dict,
12)
15class AmiCrawler(MatchingCrawler):
16 source_name = "Annales Mathematica et Informaticae website"
17 source_domain = "AMI"
18 source_website = "https://ami.uni-eszterhazy.hu/"
20 issue_re = r"Vol. \d+ \((?P<year>\d+)\)"
21 pages_re = r"Pages: (?P<fpage>\d+)–(?P<lpage>\d+)"
23 def parse_collection_content(self, content):
24 xissues = []
25 soup = BeautifulSoup(content, "html.parser")
26 issues = soup.select("#realtart select[name='vol'] option")
27 for issue in issues:
28 vol_number = issue.get("value")
29 if not isinstance(vol_number, str) or not vol_number.isdigit():
30 continue
31 issue_dict = regex_to_dict(
32 self.issue_re, issue.text, error_msg="Couldn't parse volume year"
33 )
34 xissues.append(
35 self.create_xissue(
36 self.collection_url + "?vol=" + vol_number,
37 int(issue_dict["year"]),
38 vol_number,
39 None,
40 )
41 )
42 return xissues
44 def parse_issue_content(self, content, xissue):
45 soup = BeautifulSoup(content, "html.parser")
46 articles = soup.select("#realtart p.cikk")
47 for index, article_tag in enumerate(articles):
48 xissue.articles.append(self.parse_ami_article(article_tag, xissue, index))
50 def parse_ami_article(self, article_tag: Tag, xissue: IssueData, index: int):
51 if not xissue.pid: 51 ↛ 52line 51 didn't jump to line 52 because the condition on line 51 was never true
52 raise ValueError("You must set xissue.pid before parsing an article")
53 if not xissue.url: 53 ↛ 54line 53 didn't jump to line 54 because the condition on line 53 was never true
54 raise ValueError("You must set xissue.url before parsing an article")
56 xarticle = create_articledata()
57 xarticle.lang = "en"
58 xarticle.pid = xissue.pid + "_a" + str(index)
60 add_source_link_to_xarticle(xarticle, xissue.url, self.source_domain)
62 # Title
63 title_tag = article_tag.select_one("a[href^='./uploads']")
64 if not title_tag: 64 ↛ 65line 64 didn't jump to line 65 because the condition on line 64 was never true
65 raise ValueError("Couldn't parse article title")
66 xarticle.title_tex = title_tag.text
68 # PDF
69 pdf_url = title_tag.get("href")
70 if not isinstance(pdf_url, str): 70 ↛ 71line 70 didn't jump to line 71 because the condition on line 70 was never true
71 raise ValueError("Couldn't parse article href")
72 pdf_url = urljoin(self.source_website, pdf_url)
73 add_pdf_link_to_xarticle(xarticle, pdf_url)
75 title_tag.decompose()
76 # DOI
77 doi_tag = article_tag.select_one("a[href^='https://doi.org']")
78 if doi_tag:
79 xarticle.doi = doi_tag.text
80 doi_tag.decompose()
82 # Pages
83 pages_tag = article_tag.select_one("span.oldal")
84 if not pages_tag: 84 ↛ 85line 84 didn't jump to line 85 because the condition on line 84 was never true
85 raise ValueError("Couldn't find pages")
86 pages_group = regex_to_dict(
87 self.pages_re, pages_tag.text, error_msg="Couldn't parse pages"
88 )
89 xarticle.fpage = pages_group["fpage"]
90 xarticle.lpage = pages_group["lpage"]
92 # Authors
93 authors = None
94 for child in article_tag.children: 94 ↛ 101line 94 didn't jump to line 101 because the loop on line 94 didn't complete
95 if not isinstance(child, str):
96 continue
97 child = cleanup_str(child)
98 if child.startswith("by"):
99 authors = child.removeprefix("by ")
100 break
101 if not authors: 101 ↛ 102line 101 didn't jump to line 102 because the condition on line 101 was never true
102 raise ValueError("Couldn't find authors")
104 authors = authors.split(", ")
105 for a in authors:
106 xarticle.contributors.append(create_contributor(string_name=a, role="author"))
108 return xarticle