Coverage for src/crawler/by_source/sasa_crawler.py: 83%

143 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1import re 

2from urllib.parse import urljoin 

3 

4import regex 

5from bs4 import BeautifulSoup, Tag 

6from lingua import Language, LanguageDetectorBuilder 

7from ptf.cmds.xml.xml_utils import escape 

8from ptf.model_data import ( 

9 ArticleData, 

10 IssueData, 

11 create_abstract, 

12 create_articledata, 

13 create_contributor, 

14 create_extid, 

15 create_extlink, 

16 create_subj, 

17) 

18 

19from crawler.abstract_crawlers.base_crawler import BaseCollectionCrawler 

20from crawler.utils import add_pdf_link_to_xarticle, cleanup_str 

21 

22 

23class SasaCrawler(BaseCollectionCrawler): 

24 source_name = "eLibrary of Mathematical Institute of the Serbian Academy of Sciences and Arts" 

25 source_domain = "SASA" 

26 source_website = "http://elib.mi.sanu.ac.rs" 

27 

28 _language_detector_builder = LanguageDetectorBuilder.from_languages( 

29 Language.ENGLISH, Language.SERBIAN 

30 ) 

31 

32 def parse_collection_content(self, content): 

33 soup = BeautifulSoup(content, "html.parser") 

34 xissues: list[IssueData] = [] 

35 

36 # Extract the list of issues 

37 # Filter out empty table cells 

38 volume_nodes = [ 

39 node for node in soup.select("td.issue_cell a.main_link") if node.text != "\xa0" 

40 ] 

41 for vol_node in volume_nodes: 

42 # NOTE : should we parse year here or in the issue itself ? 

43 

44 href = self.get_str_attr(vol_node, "href") 

45 

46 # Parse Volume and Issue numbers 

47 url = self.source_website + "/pages/" + href 

48 # Formats like 44_1 / 2024 | Tom XIV / 2024 | Knj. 8 / 1960 | LXIX_1-2 / 2024 

49 volume_re = list( 

50 re.finditer( 

51 r"(?P<volume>[a-zA-Z0-9 .-]+)(?:_(?P<issue>[\w-]+))? \/ (?P<year>\d+)", 

52 vol_node.text, 

53 ) 

54 ) 

55 if len(volume_re) == 0: 55 ↛ 57line 55 didn't jump to line 57 because the condition on line 55 was never true

56 # Formats like 20(28) / 2022 | 44 (1) / 2024 | (N.S.) 115 (129) / 2024 | 

57 volume_re = list( 

58 re.finditer( 

59 r"(?P<volume>[\.\( \)\w]+)\((?P<issue>\d+)\) \/ (?P<year>\d+)", 

60 vol_node.text, 

61 ) 

62 ) 

63 if len(volume_re) == 0: 63 ↛ 64line 63 didn't jump to line 64 because the condition on line 63 was never true

64 raise IndexError( 

65 f"[{self.source_domain}] {self.collection_id} : Volume cannot be parsed" 

66 ) 

67 volume_metadata = volume_re[0].groupdict() 

68 

69 # HACK : temporary workaround 

70 # https://gricad-gitlab.univ-grenoble-alpes.fr/mathdoc/ptfs/ptf-app-crawler/-/issues/27 

71 if url != "http://elib.mi.sanu.ac.rs/pages/browse_issue.php?db=flmt&rbr=95": 

72 xissues.append( 

73 self.create_xissue( 

74 url, 

75 volume_metadata["year"], 

76 volume_metadata["volume"].strip(), 

77 volume_metadata.get("issue", None), 

78 ) 

79 ) 

80 

81 # Handle pagination 

82 pages_node = soup.select_one(".page_selector") 

83 if pages_node is None: 83 ↛ 84line 83 didn't jump to line 84 because the condition on line 83 was never true

84 return xissues 

85 next_page_node = pages_node.select_one(".page_selector_dead_link+a.page_selector_link") 

86 if next_page_node is None: 

87 return xissues 

88 

89 next_page_href = self.get_str_attr(next_page_node, "href") 

90 

91 content = self.download_file(self.source_website + "/" + next_page_href) 

92 return xissues + self.parse_collection_content(content) 

93 

94 def parse_issue_content(self, content, xissue: IssueData, index: int = 0): 

95 soup = BeautifulSoup(content, "html.parser") 

96 article_nodes = soup.select(".content .result") 

97 if xissue.pid is None: 97 ↛ 98line 97 didn't jump to line 98 because the condition on line 97 was never true

98 raise ValueError( 

99 f"Error in crawler : {self.source_domain} : you must set an issue PID before parsing it" 

100 ) 

101 

102 # NOTE : publishers aren't implemented yet in base_crawler, but this should work for SASA. 

103 # issue_publisher_node = soup.select_one(".content>table td.text_cell span.data_text") 

104 # if (issue_publisher_node is not None): 

105 # publisher = issue_publisher_node.text 

106 # xpub = create_publisherdata() 

107 # xpub.name = publisher.removeprefix("Publisher ") 

108 # xissue.publisher = xpub 

109 

110 for i, art_node in enumerate(article_nodes): 

111 xarticle = self.parse_sasa_article(i + index, art_node, xissue) 

112 xissue.articles.append(xarticle) 

113 

114 index = index + len(article_nodes) 

115 # Handle pagination 

116 pages_node = soup.select_one(".page_selector") 

117 if pages_node is None: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true

118 return 

119 next_page_node = pages_node.select_one(".page_selector_dead_link+a.page_selector_link") 

120 if next_page_node is None: 

121 return 

122 

123 next_page_href = self.get_str_attr(next_page_node, "href") 

124 

125 content = self.download_file(self.source_website + "/" + next_page_href) 

126 self.parse_issue_content(content, xissue, index) 

127 

128 def parse_sasa_article( 

129 self, article_index: int, article_node: Tag, xissue: IssueData 

130 ) -> ArticleData: 

131 """ 

132 Since Sasa doesn't have pages per articles, we parse the article data from the issue page instead 

133 """ 

134 

135 title_node = article_node.select_one(".main_link") 

136 

137 if title_node is None: 137 ↛ 138line 137 didn't jump to line 138 because the condition on line 137 was never true

138 raise ValueError(f"[{self.source_domain}] {xissue.pid} : Title not found") 

139 href = self.get_str_attr(title_node, "href") 

140 

141 xarticle = create_articledata() 

142 

143 pages_node = article_node.select_one(".pages") 

144 if pages_node is not None: 144 ↛ 146line 144 didn't jump to line 146 because the condition on line 144 was always true

145 self.set_pages(xarticle, pages_node.text) 

146 xarticle.title_tex = title_node.text 

147 xarticle.title_html = title_node.text 

148 xarticle.pid = f"{xissue.pid}_a{article_index}" 

149 

150 if xissue.url is not None: 150 ↛ 158line 150 didn't jump to line 158 because the condition on line 150 was always true

151 ext_link = create_extlink( 

152 rel="source", location=xissue.url, metadata=self.source_domain 

153 ) 

154 xarticle.ext_links.append(ext_link) 

155 # xarticle.url = xissue.url 

156 

157 # Abstract 

158 abstract_node = article_node.select_one(".secondary_link:-soup-contains-own('Abstract')") 

159 

160 if abstract_node is None: 160 ↛ 161line 160 didn't jump to line 161 because the condition on line 160 was never true

161 self.logger.debug("Abstract not found", extra={"pid": xarticle.pid}) 

162 else: 

163 abstract_href = abstract_node.get("href") 

164 if abstract_href is None or isinstance(abstract_href, list): 164 ↛ 165line 164 didn't jump to line 165 because the condition on line 164 was never true

165 raise ValueError( 

166 f"[{self.source_domain}] {xarticle.pid} : Abstract href not found" 

167 ) 

168 

169 abstract = self.fetch_sasa_abstract( 

170 urljoin(self.source_website, abstract_href), xarticle.pid 

171 ) 

172 if abstract is not None: 172 ↛ 177line 172 didn't jump to line 177 because the condition on line 172 was always true

173 xarticle.abstracts.append(abstract) 

174 # LANG 

175 xarticle.lang = abstract["lang"] 

176 

177 author_node = article_node.select_one(".secondary_text") 

178 if author_node is not None: 178 ↛ 186line 178 didn't jump to line 186 because the condition on line 178 was always true

179 authors = re.findall( 

180 r"(?: and )?((?:(?<!,)(?<! and)[\w. -](?!and ))+)", author_node.text 

181 ) 

182 for a in authors: 

183 author = create_contributor(role="author", string_name=a) 

184 xarticle.contributors.append(author) 

185 else: 

186 self.logger.debug("Author not found", extra={"pid": xarticle.pid}) 

187 

188 secondary_nodes = article_node.select(".secondary_info_text") 

189 subjects = [] 

190 keywords = [] 

191 doi = None 

192 for node in secondary_nodes: 

193 text = node.text 

194 if text.startswith("Keywords"): 

195 keywords = text.removeprefix("Keywords:\xa0").split("; ") 

196 for kwd in keywords: 

197 subject = create_subj(value=kwd, lang=xarticle.lang) 

198 xarticle.kwds.append(subject) 

199 elif text.startswith("DOI") and self.collection_id != "YJOR": 

200 doi = regex.sub(r"DOI:?\s", "", text) 

201 if doi is not None: 201 ↛ 192line 201 didn't jump to line 192 because the condition on line 201 was always true

202 doi = cleanup_str(escape(doi)) 

203 # Fix for badly formatted SASA Doi 

204 # http://elib.mi.sanu.ac.rs/pages/browse_issue.php?db=kjm&rbr= 

205 if regex.match(r"(?P<doi>10[0-9]{4,}.+)", doi): 205 ↛ 206line 205 didn't jump to line 206 because the condition on line 205 was never true

206 doi = doi[:2] + "." + doi[2:] 

207 doi = doi.removeprefix("https://doi.org/") 

208 xarticle.doi = doi 

209 xarticle.pid = doi.replace("/", "_").replace(".", "_").replace("-", "_") 

210 elif text.startswith("MSC"): 

211 subjects = text.removeprefix("MSC:\xa0").split("; ") 

212 for subj in subjects: 

213 subject = create_subj(value=subj, type="msc", lang=xarticle.lang) 

214 xarticle.kwds.append(subject) 

215 elif text.startswith("Zbl:"): 215 ↛ 192line 215 didn't jump to line 192 because the condition on line 215 was always true

216 zbl_link = node.select_one(".secondary_link") 

217 if zbl_link is not None: 217 ↛ 192line 217 didn't jump to line 192 because the condition on line 217 was always true

218 extid = create_extid("zbl-item-id", zbl_link.text) 

219 xarticle.extids.append(extid) 

220 

221 if href.startswith("http"): 

222 pdf_url = href 

223 else: 

224 pdf_url = self.source_website + "/files/" + href 

225 

226 # Fix for Filomat 

227 if "www.pmf.ni.ac.rs" in pdf_url: 

228 pdf_url = pdf_url.replace("www.pmf.ni.ac.rs", "www1.pmf.ni.ac.rs") 

229 

230 add_pdf_link_to_xarticle(xarticle, pdf_url) 

231 return xarticle 

232 

233 def fetch_sasa_abstract(self, abstract_url: str, pid: str): 

234 content = self.download_file(abstract_url) 

235 soup = BeautifulSoup(content, "html.parser") 

236 text_node = soup.select_one("p") 

237 if text_node is not None: 237 ↛ 243line 237 didn't jump to line 243 because the condition on line 237 was always true

238 text = text_node.text.replace("$$", "$") 

239 abstract = create_abstract( 

240 value_tex=text, 

241 ) 

242 return abstract 

243 self.logger.debug("Abstract page exists, but text not found", extra={"pid": pid}) 

244 

245 def decode_response(self, response, encoding=None): 

246 """Attempt to decode content first before 

247 

248 SASA abstracts are encoded in windows-1250 despite the header and meta tag advertising otherwise. 

249 # example : http://elib.mi.sanu.ac.rs/files/journals/bltn/26/1e.htm 

250 """ 

251 # Attempt to get encoding using HTML meta charset tag 

252 soup = BeautifulSoup(response.text, "html5lib") 

253 charset = soup.select_one("meta[charset]") 

254 if charset: 254 ↛ 255line 254 didn't jump to line 255 because the condition on line 254 was never true

255 htmlencoding = charset.get("charset") 

256 if isinstance(htmlencoding, str): 

257 response.encoding = htmlencoding 

258 return response.text 

259 

260 return super().decode_response(response, encoding)