Coverage for src/crawler/by_source/sasa_crawler.py: 83%
143 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import re
2from urllib.parse import urljoin
4import regex
5from bs4 import BeautifulSoup, Tag
6from lingua import Language, LanguageDetectorBuilder
7from ptf.cmds.xml.xml_utils import escape
8from ptf.model_data import (
9 ArticleData,
10 IssueData,
11 create_abstract,
12 create_articledata,
13 create_contributor,
14 create_extid,
15 create_extlink,
16 create_subj,
17)
19from crawler.abstract_crawlers.base_crawler import BaseCollectionCrawler
20from crawler.utils import add_pdf_link_to_xarticle, cleanup_str
23class SasaCrawler(BaseCollectionCrawler):
24 source_name = "eLibrary of Mathematical Institute of the Serbian Academy of Sciences and Arts"
25 source_domain = "SASA"
26 source_website = "http://elib.mi.sanu.ac.rs"
28 _language_detector_builder = LanguageDetectorBuilder.from_languages(
29 Language.ENGLISH, Language.SERBIAN
30 )
32 def parse_collection_content(self, content):
33 soup = BeautifulSoup(content, "html.parser")
34 xissues: list[IssueData] = []
36 # Extract the list of issues
37 # Filter out empty table cells
38 volume_nodes = [
39 node for node in soup.select("td.issue_cell a.main_link") if node.text != "\xa0"
40 ]
41 for vol_node in volume_nodes:
42 # NOTE : should we parse year here or in the issue itself ?
44 href = self.get_str_attr(vol_node, "href")
46 # Parse Volume and Issue numbers
47 url = self.source_website + "/pages/" + href
48 # Formats like 44_1 / 2024 | Tom XIV / 2024 | Knj. 8 / 1960 | LXIX_1-2 / 2024
49 volume_re = list(
50 re.finditer(
51 r"(?P<volume>[a-zA-Z0-9 .-]+)(?:_(?P<issue>[\w-]+))? \/ (?P<year>\d+)",
52 vol_node.text,
53 )
54 )
55 if len(volume_re) == 0: 55 ↛ 57line 55 didn't jump to line 57 because the condition on line 55 was never true
56 # Formats like 20(28) / 2022 | 44 (1) / 2024 | (N.S.) 115 (129) / 2024 |
57 volume_re = list(
58 re.finditer(
59 r"(?P<volume>[\.\( \)\w]+)\((?P<issue>\d+)\) \/ (?P<year>\d+)",
60 vol_node.text,
61 )
62 )
63 if len(volume_re) == 0: 63 ↛ 64line 63 didn't jump to line 64 because the condition on line 63 was never true
64 raise IndexError(
65 f"[{self.source_domain}] {self.collection_id} : Volume cannot be parsed"
66 )
67 volume_metadata = volume_re[0].groupdict()
69 # HACK : temporary workaround
70 # https://gricad-gitlab.univ-grenoble-alpes.fr/mathdoc/ptfs/ptf-app-crawler/-/issues/27
71 if url != "http://elib.mi.sanu.ac.rs/pages/browse_issue.php?db=flmt&rbr=95":
72 xissues.append(
73 self.create_xissue(
74 url,
75 volume_metadata["year"],
76 volume_metadata["volume"].strip(),
77 volume_metadata.get("issue", None),
78 )
79 )
81 # Handle pagination
82 pages_node = soup.select_one(".page_selector")
83 if pages_node is None: 83 ↛ 84line 83 didn't jump to line 84 because the condition on line 83 was never true
84 return xissues
85 next_page_node = pages_node.select_one(".page_selector_dead_link+a.page_selector_link")
86 if next_page_node is None:
87 return xissues
89 next_page_href = self.get_str_attr(next_page_node, "href")
91 content = self.download_file(self.source_website + "/" + next_page_href)
92 return xissues + self.parse_collection_content(content)
94 def parse_issue_content(self, content, xissue: IssueData, index: int = 0):
95 soup = BeautifulSoup(content, "html.parser")
96 article_nodes = soup.select(".content .result")
97 if xissue.pid is None: 97 ↛ 98line 97 didn't jump to line 98 because the condition on line 97 was never true
98 raise ValueError(
99 f"Error in crawler : {self.source_domain} : you must set an issue PID before parsing it"
100 )
102 # NOTE : publishers aren't implemented yet in base_crawler, but this should work for SASA.
103 # issue_publisher_node = soup.select_one(".content>table td.text_cell span.data_text")
104 # if (issue_publisher_node is not None):
105 # publisher = issue_publisher_node.text
106 # xpub = create_publisherdata()
107 # xpub.name = publisher.removeprefix("Publisher ")
108 # xissue.publisher = xpub
110 for i, art_node in enumerate(article_nodes):
111 xarticle = self.parse_sasa_article(i + index, art_node, xissue)
112 xissue.articles.append(xarticle)
114 index = index + len(article_nodes)
115 # Handle pagination
116 pages_node = soup.select_one(".page_selector")
117 if pages_node is None: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true
118 return
119 next_page_node = pages_node.select_one(".page_selector_dead_link+a.page_selector_link")
120 if next_page_node is None:
121 return
123 next_page_href = self.get_str_attr(next_page_node, "href")
125 content = self.download_file(self.source_website + "/" + next_page_href)
126 self.parse_issue_content(content, xissue, index)
128 def parse_sasa_article(
129 self, article_index: int, article_node: Tag, xissue: IssueData
130 ) -> ArticleData:
131 """
132 Since Sasa doesn't have pages per articles, we parse the article data from the issue page instead
133 """
135 title_node = article_node.select_one(".main_link")
137 if title_node is None: 137 ↛ 138line 137 didn't jump to line 138 because the condition on line 137 was never true
138 raise ValueError(f"[{self.source_domain}] {xissue.pid} : Title not found")
139 href = self.get_str_attr(title_node, "href")
141 xarticle = create_articledata()
143 pages_node = article_node.select_one(".pages")
144 if pages_node is not None: 144 ↛ 146line 144 didn't jump to line 146 because the condition on line 144 was always true
145 self.set_pages(xarticle, pages_node.text)
146 xarticle.title_tex = title_node.text
147 xarticle.title_html = title_node.text
148 xarticle.pid = f"{xissue.pid}_a{article_index}"
150 if xissue.url is not None: 150 ↛ 158line 150 didn't jump to line 158 because the condition on line 150 was always true
151 ext_link = create_extlink(
152 rel="source", location=xissue.url, metadata=self.source_domain
153 )
154 xarticle.ext_links.append(ext_link)
155 # xarticle.url = xissue.url
157 # Abstract
158 abstract_node = article_node.select_one(".secondary_link:-soup-contains-own('Abstract')")
160 if abstract_node is None: 160 ↛ 161line 160 didn't jump to line 161 because the condition on line 160 was never true
161 self.logger.debug("Abstract not found", extra={"pid": xarticle.pid})
162 else:
163 abstract_href = abstract_node.get("href")
164 if abstract_href is None or isinstance(abstract_href, list): 164 ↛ 165line 164 didn't jump to line 165 because the condition on line 164 was never true
165 raise ValueError(
166 f"[{self.source_domain}] {xarticle.pid} : Abstract href not found"
167 )
169 abstract = self.fetch_sasa_abstract(
170 urljoin(self.source_website, abstract_href), xarticle.pid
171 )
172 if abstract is not None: 172 ↛ 177line 172 didn't jump to line 177 because the condition on line 172 was always true
173 xarticle.abstracts.append(abstract)
174 # LANG
175 xarticle.lang = abstract["lang"]
177 author_node = article_node.select_one(".secondary_text")
178 if author_node is not None: 178 ↛ 186line 178 didn't jump to line 186 because the condition on line 178 was always true
179 authors = re.findall(
180 r"(?: and )?((?:(?<!,)(?<! and)[\w. -](?!and ))+)", author_node.text
181 )
182 for a in authors:
183 author = create_contributor(role="author", string_name=a)
184 xarticle.contributors.append(author)
185 else:
186 self.logger.debug("Author not found", extra={"pid": xarticle.pid})
188 secondary_nodes = article_node.select(".secondary_info_text")
189 subjects = []
190 keywords = []
191 doi = None
192 for node in secondary_nodes:
193 text = node.text
194 if text.startswith("Keywords"):
195 keywords = text.removeprefix("Keywords:\xa0").split("; ")
196 for kwd in keywords:
197 subject = create_subj(value=kwd, lang=xarticle.lang)
198 xarticle.kwds.append(subject)
199 elif text.startswith("DOI") and self.collection_id != "YJOR":
200 doi = regex.sub(r"DOI:?\s", "", text)
201 if doi is not None: 201 ↛ 192line 201 didn't jump to line 192 because the condition on line 201 was always true
202 doi = cleanup_str(escape(doi))
203 # Fix for badly formatted SASA Doi
204 # http://elib.mi.sanu.ac.rs/pages/browse_issue.php?db=kjm&rbr=
205 if regex.match(r"(?P<doi>10[0-9]{4,}.+)", doi): 205 ↛ 206line 205 didn't jump to line 206 because the condition on line 205 was never true
206 doi = doi[:2] + "." + doi[2:]
207 doi = doi.removeprefix("https://doi.org/")
208 xarticle.doi = doi
209 xarticle.pid = doi.replace("/", "_").replace(".", "_").replace("-", "_")
210 elif text.startswith("MSC"):
211 subjects = text.removeprefix("MSC:\xa0").split("; ")
212 for subj in subjects:
213 subject = create_subj(value=subj, type="msc", lang=xarticle.lang)
214 xarticle.kwds.append(subject)
215 elif text.startswith("Zbl:"): 215 ↛ 192line 215 didn't jump to line 192 because the condition on line 215 was always true
216 zbl_link = node.select_one(".secondary_link")
217 if zbl_link is not None: 217 ↛ 192line 217 didn't jump to line 192 because the condition on line 217 was always true
218 extid = create_extid("zbl-item-id", zbl_link.text)
219 xarticle.extids.append(extid)
221 if href.startswith("http"):
222 pdf_url = href
223 else:
224 pdf_url = self.source_website + "/files/" + href
226 # Fix for Filomat
227 if "www.pmf.ni.ac.rs" in pdf_url:
228 pdf_url = pdf_url.replace("www.pmf.ni.ac.rs", "www1.pmf.ni.ac.rs")
230 add_pdf_link_to_xarticle(xarticle, pdf_url)
231 return xarticle
233 def fetch_sasa_abstract(self, abstract_url: str, pid: str):
234 content = self.download_file(abstract_url)
235 soup = BeautifulSoup(content, "html.parser")
236 text_node = soup.select_one("p")
237 if text_node is not None: 237 ↛ 243line 237 didn't jump to line 243 because the condition on line 237 was always true
238 text = text_node.text.replace("$$", "$")
239 abstract = create_abstract(
240 value_tex=text,
241 )
242 return abstract
243 self.logger.debug("Abstract page exists, but text not found", extra={"pid": pid})
245 def decode_response(self, response, encoding=None):
246 """Attempt to decode content first before
248 SASA abstracts are encoded in windows-1250 despite the header and meta tag advertising otherwise.
249 # example : http://elib.mi.sanu.ac.rs/files/journals/bltn/26/1e.htm
250 """
251 # Attempt to get encoding using HTML meta charset tag
252 soup = BeautifulSoup(response.text, "html5lib")
253 charset = soup.select_one("meta[charset]")
254 if charset: 254 ↛ 255line 254 didn't jump to line 255 because the condition on line 254 was never true
255 htmlencoding = charset.get("charset")
256 if isinstance(htmlencoding, str):
257 response.encoding = htmlencoding
258 return response.text
260 return super().decode_response(response, encoding)