Coverage for src/crawler/crawler_utils.py: 17%

210 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1# This file contains utils functions related to ArticleData or IssueData parsing and population 

2# Some of the functions present here were initially present in base_crawler but then moved here. 

3 

4 

5import logging 

6from collections.abc import Callable 

7from email.policy import EmailPolicy 

8 

9import regex 

10from bs4 import BeautifulSoup 

11from langcodes import standardize_tag 

12from ptf.cmds.xml.jats.builder.references import ( 

13 get_article_title_xml, 

14 get_author_xml, 

15 get_fpage_xml, 

16 get_lpage_xml, 

17 get_source_xml, 

18 get_year_xml, 

19) 

20from ptf.cmds.xml.jats.jats_parser import parse_mixed_citation_into_ref 

21from ptf.model_data import ( 

22 ArticleData, 

23 ContributorDict, 

24 IssueData, 

25 create_abstract, 

26 create_contributor, 

27 create_extid, 

28 create_issuedata, 

29 create_publisherdata, 

30) 

31 

32from crawler.types import CitationLiteral 

33from crawler.utils import add_pdf_link_to_xarticle, cleanup_str 

34 

35references_mapping = { 

36 "citation_title": get_article_title_xml, 

37 "citation_journal_title": get_source_xml, 

38 "citation_publication_date": get_year_xml, 

39 "citation_firstpage": get_fpage_xml, 

40 "citation_lastpage": get_lpage_xml, 

41} 

42 

43logger = logging.getLogger(__name__) 

44 

45 

46def parse_content_type_charset(content_type: str): 

47 header = EmailPolicy.header_factory("content-type", content_type) 

48 if "charset" in header.params: 

49 return header.params.get("charset") 

50 

51 

52def parse_meta_citation_reference(content: str, label=None): 

53 categories = content.split(";") 

54 

55 if len(categories) == 1: 

56 return parse_mixed_citation_into_ref(content, label=label) 

57 

58 citation_data = [c.split("=") for c in categories if "=" in c] 

59 del categories 

60 

61 xml_string = "" 

62 authors_parsed = False 

63 authors_strings = [] 

64 for data in citation_data: 

65 key = data[0].strip() 

66 citation_content = data[1] 

67 if key == "citation_author": 

68 authors_strings.append(get_author_xml(template_str=citation_content)) 

69 continue 

70 elif not authors_parsed: 

71 xml_string += ", ".join(authors_strings) 

72 authors_parsed = True 

73 

74 if key in references_mapping: 

75 xml_string += " " + references_mapping[key](citation_content) 

76 

77 return parse_mixed_citation_into_ref(xml_string, label=label) 

78 

79 

80def set_pages(article: ArticleData, pages: str, separator: str = "-"): 

81 pages_split = pages.split(separator) 

82 if len(pages_split) == 0: 82 ↛ 83line 82 didn't jump to line 83 because the condition on line 82 was never true

83 article.page_range = pages 

84 if len(pages_split) > 0: 84 ↛ exitline 84 didn't return from function 'set_pages' because the condition on line 84 was always true

85 if pages[0].isnumeric(): 85 ↛ exitline 85 didn't return from function 'set_pages' because the condition on line 85 was always true

86 article.fpage = pages_split[0] 

87 if ( 

88 len(pages_split) > 1 

89 and pages_split[0] != pages_split[1] 

90 and pages_split[1].isnumeric() 

91 ): 

92 article.lpage = pages_split[1] 

93 

94 

95def get_issue_pid( 

96 collection_id: str, 

97 year: int, 

98 volume_number: str | None = None, 

99 issue_number: str | None = None, 

100 series: str | None = None, 

101): 

102 # Replace any non-word character with an underscore 

103 pid = f"{collection_id}_{year}" 

104 if series is not None: 104 ↛ 105line 104 didn't jump to line 105 because the condition on line 104 was never true

105 pid += f"_{series}" 

106 if volume_number is not None: 106 ↛ 108line 106 didn't jump to line 108 because the condition on line 106 was always true

107 pid += f"_{volume_number}" 

108 if issue_number is not None: 108 ↛ 110line 108 didn't jump to line 110 because the condition on line 108 was always true

109 pid += f"_{issue_number}" 

110 pid = regex.sub(r"[^a-zA-Z0-9-]+", "_", cleanup_str(pid)) 

111 return pid 

112 

113 

114def create_xissue( 

115 collection_id: str, 

116 url: str | None, 

117 year: int, 

118 volume_number: str | None, 

119 issue_number: str | None = "1", 

120 vseries: str | None = None, 

121): 

122 if url is not None and url.endswith("/"): 122 ↛ 123line 122 didn't jump to line 123 because the condition on line 122 was never true

123 url = url[:-1] 

124 xissue = create_issuedata() 

125 xissue.url = url 

126 

127 xissue.pid = get_issue_pid(collection_id, year, volume_number, issue_number, vseries) 

128 

129 xissue.fyear = year 

130 

131 if volume_number is not None: 131 ↛ 134line 131 didn't jump to line 134 because the condition on line 131 was always true

132 xissue.volume = regex.sub(r"[^a-zA-Z0-9-]+", "_", volume_number) 

133 

134 if issue_number is not None: 134 ↛ 137line 134 didn't jump to line 137 because the condition on line 134 was always true

135 xissue.number = issue_number.replace(",", "-") 

136 

137 if vseries is not None: 137 ↛ 138line 137 didn't jump to line 138 because the condition on line 137 was never true

138 xissue.vseries = vseries 

139 return xissue 

140 

141 

142def get_metadata_using_citation_meta( 

143 xarticle: ArticleData, 

144 xissue: IssueData, 

145 soup: BeautifulSoup, 

146 what: list[CitationLiteral] = [], 

147 detect_language_fct: Callable[[str, ArticleData], str] | None = None, 

148): 

149 """ 

150 :param xarticle: the xarticle that will collect the metadata 

151 :param xissue: the xissue that will collect the publisher 

152 :param soup: the BeautifulSoup object of tha article page 

153 :param what: list of citation_ items to collect. 

154 :return: None. The given article is modified 

155 """ 

156 

157 if "title" in what: 

158 # TITLE 

159 citation_title_node = soup.select_one("meta[name='citation_title']") 

160 if citation_title_node: 

161 title = citation_title_node.get("content") 

162 if isinstance(title, str): 

163 xarticle.title_tex = title 

164 

165 if "author" in what: 

166 # AUTHORS 

167 citation_author_nodes = soup.select("meta[name^='citation_author']") 

168 current_author: ContributorDict | None = None 

169 for citation_author_node in citation_author_nodes: 

170 if citation_author_node.get("name") == "citation_author": 

171 text_author = citation_author_node.get("content") 

172 if not isinstance(text_author, str): 

173 raise ValueError("Cannot parse author") 

174 if text_author == "": 

175 current_author = None 

176 continue 

177 current_author = create_contributor(role="author", string_name=text_author) 

178 xarticle.contributors.append(current_author) 

179 continue 

180 if current_author is None: 

181 logger.warning("Couldn't parse citation author") 

182 continue 

183 if citation_author_node.get("name") == "citation_author_institution": 

184 text_institution = citation_author_node.get("content") 

185 if not isinstance(text_institution, str): 

186 continue 

187 current_author["addresses"].append(text_institution) 

188 if citation_author_node.get("name") == "citation_author_ocrid": 

189 text_orcid = citation_author_node.get("content") 

190 if not isinstance(text_orcid, str): 

191 continue 

192 current_author["orcid"] = text_orcid 

193 

194 if "pdf" in what: 

195 # PDF 

196 citation_pdf_node = soup.select_one('meta[name="citation_pdf_url"]') 

197 if citation_pdf_node: 

198 pdf_url = citation_pdf_node.get("content") 

199 if isinstance(pdf_url, str): 

200 add_pdf_link_to_xarticle(xarticle, pdf_url) 

201 

202 if "lang" in what: 

203 # LANG 

204 citation_lang_node = soup.select_one("meta[name='citation_language']") 

205 if citation_lang_node: 

206 # TODO: check other language code 

207 content_text = citation_lang_node.get("content") 

208 if isinstance(content_text, str): 

209 xarticle.lang = standardize_tag(content_text) 

210 

211 if "abstract" in what: 

212 # ABSTRACT 

213 abstract_node = soup.select_one("meta[name='citation_abstract']") 

214 if abstract_node is not None: 

215 abstract = abstract_node.get("content") 

216 if not isinstance(abstract, str): 

217 raise ValueError("Couldn't parse abstract from meta") 

218 abstract = BeautifulSoup(abstract, "html.parser").text 

219 lang = abstract_node.get("lang") 

220 if not isinstance(lang, str): 

221 if not detect_language_fct: 

222 return 

223 lang = detect_language_fct(abstract, xarticle) 

224 xarticle.abstracts.append(create_abstract(lang=lang, value_tex=abstract)) 

225 

226 if "page" in what: 

227 # PAGES 

228 citation_fpage_node = soup.select_one("meta[name='citation_firstpage']") 

229 if citation_fpage_node: 

230 page = citation_fpage_node.get("content") 

231 if isinstance(page, str): 

232 page = page.split("(")[0] 

233 if len(page) < 32: 

234 xarticle.fpage = page 

235 

236 citation_lpage_node = soup.select_one("meta[name='citation_lastpage']") 

237 if citation_lpage_node: 

238 page = citation_lpage_node.get("content") 

239 if isinstance(page, str): 

240 page = page.split("(")[0] 

241 if len(page) < 32: 

242 xarticle.lpage = page 

243 

244 if "doi" in what: 

245 # DOI 

246 citation_doi_node = soup.select_one("meta[name='citation_doi']") 

247 if citation_doi_node: 

248 doi = citation_doi_node.get("content") 

249 if isinstance(doi, str): 

250 doi = doi.strip() 

251 pos = doi.find("10.") 

252 if pos > 0: 

253 doi = doi[pos:] 

254 xarticle.doi = doi 

255 

256 if "mr" in what: 

257 # MR 

258 citation_mr_node = soup.select_one("meta[name='citation_mr']") 

259 if citation_mr_node: 

260 mr = citation_mr_node.get("content") 

261 if isinstance(mr, str): 

262 mr = mr.strip() 

263 if mr.find("MR") == 0: 

264 mr = mr[2:] 

265 extid = create_extid("mr-item-id", mr) 

266 xarticle.extids.append(extid) 

267 

268 if "zbl" in what: 

269 # ZBL 

270 citation_zbl_node = soup.select_one("meta[name='citation_zbl']") 

271 if citation_zbl_node: 

272 zbl = citation_zbl_node.get("content") 

273 if isinstance(zbl, str): 

274 zbl = zbl.strip() 

275 if zbl.find("Zbl") == 0: 

276 zbl = zbl[3:].strip() 

277 extid = create_extid("zbl-item-id", zbl) 

278 xarticle.extids.append(extid) 

279 

280 if "publisher" in what: 

281 # PUBLISHER 

282 citation_publisher_node = soup.select_one("meta[name='citation_publisher']") 

283 if citation_publisher_node: 

284 pub = citation_publisher_node.get("content") 

285 if isinstance(pub, str): 

286 pub = pub.strip() 

287 if pub != "": 

288 xpub = create_publisherdata() 

289 xpub.name = pub 

290 xissue.publisher = xpub 

291 

292 if "keywords" in what: 

293 # KEYWORDS 

294 citation_kwd_nodes = soup.select("meta[name='citation_keywords']") 

295 for kwd_node in citation_kwd_nodes: 

296 kwds = kwd_node.get("content") 

297 if isinstance(kwds, str): 

298 kwds = kwds.split(",") 

299 for kwd in kwds: 

300 if kwd == "": 

301 continue 

302 kwd = kwd.strip() 

303 xarticle.kwds.append({"type": "", "lang": xarticle.lang, "value": kwd}) 

304 

305 if "references" in what: 

306 citation_references = soup.select("meta[name='citation_reference']") 

307 for index, tag in enumerate(citation_references): 

308 content = tag.get("content") 

309 if not isinstance(content, str): 

310 raise ValueError("Cannot parse citation_reference meta") 

311 label = str(index + 1) 

312 if regex.match(r"^\[\d+\].*", content): 

313 label = None 

314 xarticle.bibitems.append(parse_meta_citation_reference(content, label)) 

315 

316 

317def article_has_pdf(art: ArticleData | IssueData): 

318 return next((link for link in art.ext_links if link["rel"] == "article-pdf"), None) is not None 

319 

320 

321def article_has_source(art: ArticleData | IssueData): 

322 return ( 

323 next( 

324 (e_link for e_link in art.ext_links if e_link["rel"] == "source"), 

325 None, 

326 ) 

327 is not None 

328 )