Coverage for src/crawler/by_source/dml_e_crawler.py: 28%

173 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1from urllib.parse import urljoin 

2 

3import regex 

4from bs4 import BeautifulSoup, Tag 

5from ptf.model_data import ( 

6 ArticleData, 

7 IssueData, 

8 create_articledata, 

9 create_contributor, 

10 create_extid, 

11 create_extlink, 

12) 

13 

14from crawler.abstract_crawlers.matching_crawler import MatchingCrawler 

15from crawler.crawler_utils import article_has_source 

16from crawler.models import ExtlinkChecked 

17from crawler.utils import add_pdf_link_to_xarticle 

18 

19 

20class Dml_eCrawler(MatchingCrawler): 

21 """ 

22 DML_E is quite peculiar : 

23 There is no issue page, and articles are separated into "years" instead of volumes/issues. 

24 volume/issue number is stored inside each article page. 

25 In order to being able to parse volume and issue numbers, we must parse the articles before creating volumes and issues. 

26 """ 

27 

28 source_domain = "DML_E" 

29 source_name = "Proyecto DML-E: Biblioteca Digital de Matemáticas " 

30 source_website = "http://dmle.icmat.es/" 

31 

32 # 1987, 1: 1-17 

33 # 1999,19: 1-11 

34 # 2008, 53-62, 

35 # 1963 (1-2): 

36 # 2000, 51 (1): 49-58, 13 Ref. 

37 # 2006, 57 (Extra): 327-342, 10 Ref. 

38 issue_regex = r"\d+,? ?(?:(?P<volume>\d+),? ?)?(?:\((?P<number>[\d\w\-]+)\))?(?:[:,])? ?(?:(?P<page_start>\d+)-(?P<page_end>\d+))?" 

39 

40 def parse_collection_content(self, content): 

41 xissues = [] 

42 soup = BeautifulSoup(content, "html.parser") 

43 pagination_elements = soup.select("div.prevnext a") 

44 for page in pagination_elements: 

45 href = page.get("href") 

46 if not isinstance(href, str): 46 ↛ 47line 46 didn't jump to line 47 because the condition on line 46 was never true

47 continue 

48 href = urljoin(self.collection_url, href) 

49 content = self.download_file(href) 

50 xissues = [*xissues, *self.parse_collection_page(content, href)] 

51 

52 return xissues 

53 

54 def parse_collection_page(self, content: str, url: str): 

55 soup = BeautifulSoup(content, "html.parser") 

56 xissues = [] 

57 current_year = False 

58 issues_tags = soup.select("a[name], ul.art_info") 

59 for issue_tag in issues_tags: 

60 if issue_tag.name == "a": 

61 current_year = issue_tag.get("name") 

62 if not isinstance(current_year, str): 62 ↛ 63line 62 didn't jump to line 63 because the condition on line 62 was never true

63 raise ValueError("Issue year cannot be parsed") 

64 continue 

65 

66 if not current_year: 66 ↛ 67line 66 didn't jump to line 67 because the condition on line 66 was never true

67 raise ValueError("Issue year not found") 

68 issue = self.create_xissue(url, int(current_year), current_year) 

69 self.parse_issue_tag(issue_tag, issue) 

70 xissues.append(issue) 

71 return xissues 

72 

73 # def parse_issue_content(self, content, xissue): 

74 # pass 

75 

76 def parse_issue_tag(self, tag: Tag, xissue: IssueData): 

77 if not xissue.url: 77 ↛ 78line 77 didn't jump to line 78 because the condition on line 77 was never true

78 raise ValueError("xissue must have an URL") 

79 article_tags = tag.select("li") 

80 for index, art_tag in enumerate(article_tags): 

81 href_tag = art_tag.select_one("a[href]") 

82 if not href_tag: 82 ↛ 83line 82 didn't jump to line 83 because the condition on line 82 was never true

83 raise ValueError("Cannot parse article") 

84 url = href_tag.get("href") 

85 if not isinstance(url, str): 85 ↛ 86line 85 didn't jump to line 86 because the condition on line 85 was never true

86 raise ValueError("Cannot parse Article URL") 

87 url = urljoin(xissue.url, url) 

88 

89 title = href_tag.text 

90 

91 article = create_articledata() 

92 article.title_tex = title 

93 article.url = url 

94 article.pid = "a" + str(index) 

95 xissue.articles.append(article) 

96 

97 def parse_dml_e_article_content(self, content, xissue, xarticle: ArticleData, url, pid): 

98 xarticle.pid = pid 

99 soup = BeautifulSoup(content, "html.parser") 

100 table_lines = soup.select("div#centro table tr") 

101 issue_volume: str | None = None 

102 issue_number: str | None = None 

103 for line in table_lines: 

104 header_tag = line.select_one("th") 

105 value_tag = line.select_one("td") 

106 if not value_tag: 

107 raise ValueError("Cannot parse article") 

108 

109 # PDF 

110 if not header_tag: 

111 href_tag = line.select_one("a") 

112 if not href_tag: 

113 raise ValueError("Cannot parse article pdf link") 

114 href = href_tag.get("href") 

115 if not isinstance(href, str): 

116 raise ValueError("Cannot parse article pdf link") 

117 add_pdf_link_to_xarticle(xarticle, self.source_website + href) 

118 continue 

119 

120 # Title 

121 if header_tag.text == "Título español": 

122 xarticle.title_tex = value_tag.text 

123 continue 

124 if header_tag.text == "Título original": 

125 xarticle.title_tex = value_tag.text 

126 continue 

127 if header_tag.text == "Título inglés": 

128 xarticle.title_tex = value_tag.text 

129 continue 

130 

131 # Author 

132 if header_tag.text == "Autor/es": 

133 authors_tags = value_tag.select("a") 

134 for a in authors_tags: 

135 author = create_contributor() 

136 author["role"] = "author" 

137 author["string_name"] = a.text 

138 xarticle.contributors.append(author) 

139 continue 

140 # Page 

141 if header_tag.text == "Publicación": 

142 volume_re = list(regex.finditer(self.issue_regex, value_tag.text)) 

143 if len(volume_re) != 0: 

144 # raise ValueError("Cannot parse Article page") 

145 volume_data = volume_re[0].groupdict() 

146 

147 if volume_data["page_start"] and volume_data["page_end"]: 

148 xarticle.page_range = ( 

149 volume_data["page_start"] + "-" + volume_data["page_end"] 

150 ) 

151 if "volume" in volume_data: 

152 issue_volume = volume_data["volume"] 

153 if "number" in volume_data: 

154 issue_number = volume_data["number"] 

155 else: 

156 raise ValueError("issue volume or number not found") 

157 

158 # LANG 

159 if header_tag.text == "Idioma": 

160 languages = {"Inglés": "en", "Español": "es", "Francés": "fr"} 

161 if value_tag.text in languages: 

162 xarticle.lang = languages[value_tag.text] 

163 

164 if header_tag.text == "Código MathReviews": 

165 if value_tag.text.startswith("MR"): 

166 extid = create_extid("mr-item-id", value_tag.text) 

167 xarticle.extids.append(extid) 

168 if header_tag.text == "Código Z-Math": 

169 if value_tag.text.startswith("Zbl "): 

170 # Space in zblid... http://dmle.icmat.es/revistas/detalle.php?numero=95 

171 zblid = value_tag.text.removeprefix("Zbl ").strip() 

172 # zblid does not seems to exist http://dmle.icmat.es/revistas/detalle.php?numero=4004 

173 if not zblid.startswith("pre"): 

174 extid = create_extid("zbl-item-id", zblid) 

175 xarticle.extids.append(extid) 

176 

177 return xarticle, issue_volume, issue_number 

178 

179 def crawl_issue(self, xissue: IssueData): 

180 if hasattr(xissue, "url") and xissue.url: 

181 content = self.download_file(xissue.url) 

182 self.parse_issue_content(content, xissue) 

183 

184 dml_e_issues: dict[str, IssueData] = {} 

185 

186 xarticles = xissue.articles 

187 

188 for xarticle in xarticles: 

189 parsed_xarticle, xissue_vol, xissue_number = self.crawl_dml_e_article(xarticle, xissue) 

190 if parsed_xarticle is None: 

191 continue 

192 if xissue_vol or xissue_number: 

193 issue_tag = (xissue_vol or "") + "_" + (xissue_number or "") 

194 else: 

195 issue_tag = xissue.fyear 

196 if not issue_tag: 

197 raise ValueError("issue_tag is None") 

198 if issue_tag not in dml_e_issues: 

199 dml_e_issues[issue_tag] = self.create_xissue( 

200 xissue.url, xissue.fyear, xissue_vol, xissue_number or None 

201 ) 

202 dml_e_issues[issue_tag].articles.append(parsed_xarticle) 

203 

204 for value in dml_e_issues.values(): 

205 if self.ignore_missing_pdf: 

206 value.articles = [a for a in value.articles if self.article_has_pdf(a)] 

207 if self.dry: 

208 return 

209 issue_has_pdf = self.article_has_pdf(value) 

210 if len(value.articles) == 0 and not issue_has_pdf: 

211 continue 

212 for index, article in enumerate(value.articles): 

213 article.pid = f"{value.pid}_a{index}" 

214 self.process_resource_metadata(value, resource_type="issue") 

215 self.add_xissue_into_database(value) 

216 

217 def crawl_dml_e_article(self, xarticle: ArticleData, xissue: IssueData): 

218 parsed_xarticle = xarticle 

219 if not hasattr(xarticle, "url") or not xarticle.url: 

220 raise ValueError("article does not have an url") 

221 # self.progress_bar.text(f"{xarticle.pid} - {xarticle.url}") 

222 

223 content = self.download_file(xarticle.url) 

224 pid = f"{xissue.pid}_{xarticle.pid}" 

225 

226 parsed_xarticle, xissue_vol, xissue_number = self.parse_dml_e_article_content( 

227 content, xissue, xarticle, xarticle.url, pid 

228 ) 

229 

230 if not article_has_source(parsed_xarticle) and parsed_xarticle.url: 

231 ext_link = create_extlink() 

232 ext_link["rel"] = "source" 

233 ext_link["location"] = parsed_xarticle.url 

234 ext_link["metadata"] = self.source_domain 

235 parsed_xarticle.ext_links.append(ext_link) 

236 

237 # The article title may have formulas surrounded with '$' 

238 return self.process_article_metadata(parsed_xarticle), xissue_vol, xissue_number 

239 

240 @classmethod 

241 def check_pdf_link_validity(cls, url, verify=True): 

242 # we overwrite this base_crawler method to manage the links to pdf that are not article pdf. 

243 # Avoid downloading the whole PDF 

244 # CHUNK_SIZE = 100 # number of characters fetched 

245 # If the url contains Movingwall it does not lead to the article 

246 # TODO this should be in the harvest tasks 

247 if "Movingwall" in url: 

248 print("The url does not link to the PDF article because of o moving wall") 

249 return ( 

250 False, 

251 None, 

252 { 

253 "status": ExtlinkChecked.Status.ERROR, 

254 "message": "The url does not link to the PDF article because of a moving wall", 

255 }, 

256 ) 

257 return super().check_pdf_link_validity(url)