Coverage for src/crawler/abstract_crawlers/base_crawler.py: 63%

610 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1import logging 

2import time 

3from collections.abc import Iterable 

4from datetime import datetime, timedelta 

5from email.policy import EmailPolicy 

6from typing import TYPE_CHECKING, Literal 

7 

8import aiohttp 

9import regex 

10import requests 

11from bs4 import BeautifulSoup 

12from django.conf import settings 

13from django.contrib.auth.models import User 

14from django.db.utils import IntegrityError 

15from django.utils import timezone 

16from langcodes import standardize_tag 

17from lingua import LanguageDetector, LanguageDetectorBuilder 

18from opentelemetry import trace 

19from ptf.cmds.xml.ckeditor.utils import ( 

20 build_jats_data_from_html_field, 

21) 

22from ptf.cmds.xml.jats.builder.references import ( 

23 get_article_title_xml, 

24 get_author_xml, 

25 get_fpage_xml, 

26 get_lpage_xml, 

27 get_source_xml, 

28 get_year_xml, 

29) 

30from ptf.cmds.xml.jats.jats_parser import parse_mixed_citation_into_ref 

31from ptf.external.session import get_session 

32from ptf.model_data import ( 

33 ArticleData, 

34 ContributorDict, 

35 IssueData, 

36 ResourceData, 

37 TitleDict, 

38 create_abstract, 

39 create_contributor, 

40 create_extid, 

41 create_issuedata, 

42 create_publisherdata, 

43 create_subj, 

44 create_titledata, 

45) 

46from ptf.model_data_converter import update_data_for_jats 

47from ptf.models import ExtLink 

48from pylatexenc.latex2text import LatexNodes2Text 

49from pysolr import SolrError 

50from requests_cache import CachedSession 

51 

52from crawler.cmds.xml_cmds import addOrUpdateGDMLIssueXmlCmd 

53from crawler.models import Source 

54from crawler.models.extlink_checked import ExtlinkChecked 

55from crawler.types import CitationLiteral 

56from crawler.utils import ( 

57 add_pdf_link_to_xarticle, 

58 add_source_link_to_xarticle, 

59 cleanup_str, 

60 get_all_cols, 

61 get_or_create_collection, 

62) 

63 

64if TYPE_CHECKING: 

65 from collections.abc import Callable 

66 

67 from bs4 import Tag 

68 

69 

70class CrawlerTitleDict(TitleDict): 

71 title_tex: str | None 

72 

73 

74class BaseCollectionCrawler: 

75 """ 

76 Base collection for the crawlers. 

77 To create a crawler: 

78 1) derive a class from BaseCollectionCrawler and name it XXXCrawler 

79 2) override the functions parse_collection_content, parse_issue_content and parse_article_content 

80 3) update factory.py so that crawler_factory can return your new crawler 

81 """ 

82 

83 logger = logging.getLogger(__name__) 

84 tracer = trace.get_tracer(__name__) 

85 

86 source_name = "" 

87 source_domain = "" 

88 source_website = "" 

89 

90 issue_href = "" 

91 

92 collection = None 

93 source = None 

94 user = None 

95 session: requests.Session | CachedSession 

96 async_session: aiohttp.ClientSession 

97 is_checkable = True 

98 verify = True 

99 headers = {} 

100 

101 requests_interval = getattr(settings, "REQUESTS_INTERVAL", 90) 

102 "seconds to wait between two http requests" 

103 requests_timeout = 60 

104 "seconds to wait before aborting the connection (if no bytes are recieved)" 

105 

106 latext_parser = LatexNodes2Text() 

107 

108 # Override the values in your concrete crawler if the formulas in text (titles, abstracts) 

109 # do not use the "$" to surround tex formulas 

110 delimiter_inline_formula = "$" 

111 delimiter_disp_formula = "$" 

112 

113 # HACK : Workaround for tests (monkeypatching) 

114 # We store the class here, so we can monkeypatch it when running tests 

115 # subCrawlers = { 

116 # LofplCrawler: None 

117 # } 

118 subCrawlers: dict[type["BaseCollectionCrawler"], "BaseCollectionCrawler | None"] = {} 

119 

120 _language_detector: LanguageDetector | None = None 

121 _language_detector_builder = LanguageDetectorBuilder.from_all_languages() 

122 

123 match_headers = False 

124 "Whereas to include headers in requests cache key" 

125 orcid_re = r"https\:\/\/orcid\.org\/(?P<orcid>\d{4}-\d{4}-\d{4}-\d{4})" 

126 

127 ignore_missing_pdf = True 

128 "Set this to False on a Crawler-basis to allow inserting articles without PDFs" 

129 pid_year_restrictions: dict[str, int] = {} 

130 "pid -> excluded years count" 

131 

132 pause_function: "Callable[[int], None]" 

133 "Overridable the pause function (used in celery tasks to speedup aborting)" 

134 

135 @classmethod 

136 def get_view_id(cls): 

137 return cls.source_domain 

138 

139 @property 

140 def language_detector(self): 

141 """Crawler Instance singleton for language builder. 

142 Late init of LanguageDetector to save on memory""" 

143 if not self._language_detector: 

144 self._language_detector = self._language_detector_builder.build() 

145 return self._language_detector 

146 

147 def __init__( 

148 self, 

149 *args, 

150 username: str, 

151 collection_id: str, 

152 dry: bool = False, 

153 publisher: str = "", 

154 collection_url: str | None = None, 

155 backend=None, 

156 pause_function=staticmethod(time.sleep), 

157 ): 

158 if not collection_url: 158 ↛ 159line 158 didn't jump to line 159 because the condition on line 158 was never true

159 all_cols = get_all_cols() 

160 col = all_cols[collection_id] 

161 

162 collection_url = col["sources"].get(self.source_domain, None) 

163 if collection_url is None: 

164 raise ValueError( 

165 f"Source {self.source_domain} not found for collection {collection_id}" 

166 ) 

167 self.collection_url = collection_url 

168 for CrawlerClass in self.subCrawlers: 168 ↛ 169line 168 didn't jump to line 169 because the loop on line 168 never started

169 self.subCrawlers[CrawlerClass] = CrawlerClass( 

170 *args, 

171 username=username, 

172 collection_id=collection_id, 

173 dry=dry, 

174 publisher=publisher, 

175 collection_url=collection_url, 

176 ) 

177 self.logger = logging.getLogger(__name__ + "." + self.source_domain) 

178 # self.logger = logging.getLogger(__name__) 

179 

180 self.username = username 

181 

182 self.collection_id = collection_id 

183 

184 self.dry = dry 

185 self.publisher = publisher 

186 

187 # Classproperty : We sometimes want to use the session without initializing the class (rot monitoring) 

188 BaseCollectionCrawler.session = requests.Session() 

189 

190 self.pause_function = pause_function 

191 

192 # Skipped when running tests 

193 self.initialize() 

194 

195 self.backend = backend 

196 

197 def initialize(self): 

198 """ 

199 Acts as a "second" init function to skip model accesses during test data generation 

200 """ 

201 self.collection = get_or_create_collection(self.collection_id) 

202 self.source = self.get_or_create_source() 

203 self.user = User.objects.get(username=self.username) 

204 BaseCollectionCrawler.session = get_session() 

205 BaseCollectionCrawler.session.verify = self.verify 

206 self.session.pause_function = self.pause_function 

207 self.session.delay = self.requests_interval 

208 

209 @classmethod 

210 def can_crawl(cls, pid: str) -> bool: 

211 return True 

212 

213 def parse_collection_content(self, content: str) -> list[IssueData]: 

214 """ 

215 Parse the HTML content with BeautifulSoup 

216 returns a list of xissue. 

217 Override this function in a derived class 

218 """ 

219 return [] 

220 

221 def parse_issue_content(self, content: str, xissue: IssueData): 

222 """ 

223 Parse the HTML content with BeautifulSoup 

224 Fills the xissue.articles 

225 Override this function in a derived class. 

226 

227 CAV : You are supposed to create articles there. Please assign a PID to each article. 

228 The PID can be `a + article_index`, like this : `a0` `a21` 

229 """ 

230 

231 def parse_article_content( 

232 self, content: str, xissue: IssueData, xarticle: ArticleData, url: str 

233 ) -> ArticleData | None: 

234 """ 

235 Parse the HTML content with BeautifulSoup 

236 returns the xarticle. 

237 Override this function in a derived class. 

238 The xissue is passed to the function in case the article page has issue information (ex: publisher) 

239 The article url is also passed as a parameter 

240 

241 CAV : You are supposed to assign articles pid again here 

242 """ 

243 return xarticle 

244 

245 @tracer.start_as_current_span("crawl_collection") 

246 def crawl_collection(self): 

247 # TODO: Comments, filter 

248 """ 

249 Crawl an entire collection. ptf.models.Container objects are created. 

250 - get the HTML content of the collection_url 

251 - parse the HTML content with beautifulsoup to extract the list of issues 

252 - merge the xissues (some Source can have multiple pages for 1 volume/issue. We create only 1 container) 

253 - crawl each issue if col_only is False 

254 - Returns the list of merged issues. 

255 It is an OrderedDict {pid: {"issues": xissues}} 

256 The key is the pid of the merged issues. 

257 Ex: The source may have Ex: Volume 6 (2000) and Volume 6 (1999) 

258 the pid is then made with 1999-2000__6_ 

259 """ 

260 

261 if self.source is None: 

262 raise RuntimeError("ERROR: the source is not set") 

263 

264 content = self.download_file(self.collection_url) 

265 if content: 

266 xissues = self.parse_collection_content(content) 

267 else: 

268 # download_file returns None (404) 

269 return None 

270 

271 """ 

272 Some collections split the same volumes in different pages 

273 Ex: Volume 6 (2000) and Volume 6 (1999) 

274 We merge the 2 xissues with the same volume number => Volume 6 (1999-2000) 

275 """ 

276 # merged_xissues = self.merge_xissues(xissues) 

277 

278 xissues_dict = {str(i.pid): i for i in xissues} 

279 

280 return xissues_dict 

281 

282 def start_process_issue(self, xissue: IssueData): 

283 # Some source, like EuDML do not have a separate HTML pages for an issue's table of content. 

284 # The list of articles directly come from the collection HTML page: the xissue has no url attribute 

285 issue_url = xissue.url 

286 if issue_url is not None: 

287 if issue_url.endswith(".pdf"): 

288 add_pdf_link_to_xarticle(xissue, issue_url) 

289 xissue.url = None 

290 else: 

291 content = self.download_file(issue_url) 

292 with self.tracer.start_as_current_span("parse_issue_content"): 

293 self.parse_issue_content(content, xissue) 

294 

295 @tracer.start_as_current_span("crawl_issue") 

296 def crawl_issue(self, xissue: IssueData): 

297 """ 

298 Crawl 1 wag page of an issue. 

299 - get the HTML content of the issue 

300 - parse the HTML content with beautifulsoup to extract the list of articles and/or the issue metadata 

301 - crawl each article 

302 """ 

303 

304 self.start_process_issue(xissue) 

305 

306 xarticles = xissue.articles 

307 

308 parsed_xarticles = [] 

309 

310 for xarticle in xarticles: 

311 parsed_xarticle = self.crawl_article(xarticle, xissue) 

312 if parsed_xarticle is not None: 

313 parsed_xarticles.append(parsed_xarticle) 

314 

315 xissue.articles = parsed_xarticles 

316 

317 issue_has_pdf = self.article_has_pdf(xissue) 

318 

319 if self.ignore_missing_pdf: 

320 xissue.articles = [a for a in xissue.articles if self.article_has_pdf(a)] 

321 if self.dry: 

322 return 

323 if len(xissue.articles) == 0 and not issue_has_pdf: 

324 return 

325 self.process_resource_metadata(xissue, resource_type="issue") 

326 

327 self.add_xissue_into_database(xissue) 

328 

329 @staticmethod 

330 def article_has_source(art: ArticleData | IssueData): 

331 return ( 

332 next( 

333 (e_link for e_link in art.ext_links if e_link["rel"] == "source"), 

334 None, 

335 ) 

336 is not None 

337 ) 

338 

339 @staticmethod 

340 def article_has_pdf(art: ArticleData | IssueData): 

341 return ( 

342 next( 

343 (link for link in art.ext_links if link["rel"] in ["article-pdf", "article-html"]), 

344 None, 

345 ) 

346 is not None 

347 ) 

348 

349 def crawl_article(self, xarticle: ArticleData, xissue: IssueData): 

350 # ARTICLE URL as en ExtLink (to display the link in the article page) 

351 if xarticle.url is None: 

352 if not self.article_has_source(xarticle): 352 ↛ 358line 352 didn't jump to line 358 because the condition on line 352 was always true

353 if xissue.url: 

354 article_source = xissue.url 

355 else: 

356 article_source = self.collection_url 

357 add_source_link_to_xarticle(xarticle, article_source, self.source_domain) 

358 return self.process_article_metadata(xarticle) 

359 

360 if self.parse_article_content.__func__ != BaseCollectionCrawler.parse_article_content: 

361 content = self.download_file(xarticle.url) 

362 xarticle.pid = f"{xissue.pid}_{xarticle.pid}" 

363 

364 try: 

365 with self.tracer.start_as_current_span("parse_article_content"): 

366 parsed_xarticle = self.parse_article_content( 

367 content, xissue, xarticle, xarticle.url 

368 ) 

369 except ValueError as e: 

370 self.logger.warning(e) 

371 self.logger.warning("Retrying in 5 mins while invalidating cache") 

372 self.pause_function(5 * 60) 

373 content = self.download_file(xarticle.url, headers={"Cache-Control": "no-cache"}) 

374 with self.tracer.start_as_current_span("parse_article_content"): 

375 parsed_xarticle = self.parse_article_content( 

376 content, xissue, xarticle, xarticle.url 

377 ) 

378 

379 if not parsed_xarticle: 379 ↛ 380line 379 didn't jump to line 380 because the condition on line 379 was never true

380 return None 

381 

382 xarticle = parsed_xarticle 

383 

384 if xarticle.doi: 

385 xarticle.pid = xarticle.doi.replace("/", "_").replace(".", "_").replace("-", "_") 

386 

387 if not self.article_has_source(xarticle) and xarticle.url: 

388 add_source_link_to_xarticle(xarticle, xarticle.url, self.source_domain) 

389 

390 # The article title may have formulas surrounded with '$' 

391 return self.process_article_metadata(xarticle) 

392 

393 def process_resource_metadata(self, xresource: ResourceData, resource_type="article"): 

394 tag = "article-title" if resource_type == "article" else "issue-title" 

395 

396 # Process title tex 

397 ckeditor_data = build_jats_data_from_html_field( 

398 xresource.title_tex, 

399 tag=tag, 

400 text_lang=xresource.lang, 

401 delimiter_inline=self.delimiter_inline_formula, 

402 delimiter_disp=self.delimiter_disp_formula, 

403 ) 

404 

405 xresource.title_html = ckeditor_data["value_html"] 

406 # xresource.title_tex = ckeditor_data["value_tex"] 

407 xresource.title_xml = ckeditor_data["value_xml"] 

408 

409 abstracts_to_parse = [ 

410 xabstract for xabstract in xresource.abstracts if xabstract["tag"] == "abstract" 

411 ] 

412 # abstract may have formulas surrounded with '$' 

413 if len(abstracts_to_parse) > 0: 

414 for xabstract in abstracts_to_parse: 

415 ckeditor_data = build_jats_data_from_html_field( 

416 xabstract["value_tex"], 

417 tag="abstract", 

418 text_lang=xabstract["lang"], 

419 resource_lang=xresource.lang, 

420 field_type="abstract", 

421 delimiter_inline=self.delimiter_inline_formula, 

422 delimiter_disp=self.delimiter_disp_formula, 

423 ) 

424 

425 xabstract["value_html"] = ckeditor_data["value_html"] 

426 # xabstract["value_tex"] = ckeditor_data["value_tex"] 

427 xabstract["value_xml"] = ckeditor_data["value_xml"] 

428 

429 # Remove useless titles (duplicates) 

430 titles = [] 

431 for trans_title in xresource.titles: 

432 if trans_title["title_html"] != xresource.title_html: 432 ↛ 431line 432 didn't jump to line 431 because the condition on line 432 was always true

433 # trans_title["title_html"] = self.latex_converter.latex_to_text( 

434 # trans_title["title_html"] 

435 # ) 

436 titles.append(trans_title) 

437 

438 xresource.titles = titles 

439 

440 # remove duplicated keywords and add "msc" type to msc keywords : 

441 xresource_keywords = [] 

442 for resource_keyword_dict in xresource.kwds: 

443 resource_keyword = resource_keyword_dict["value"] 

444 if resource_keyword == "" or resource_keyword in [ 444 ↛ 447line 444 didn't jump to line 447 because the condition on line 444 was never true

445 xresource_keyword["value"] for xresource_keyword in xresource_keywords 

446 ]: 

447 continue 

448 

449 if any(char.isdigit() for char in resource_keyword): 

450 resource_keyword_dict["type"] = "msc" 

451 

452 xresource_keywords.append(resource_keyword_dict) 

453 xresource.kwds = xresource_keywords 

454 

455 return xresource 

456 

457 def process_article_metadata(self, xarticle: ArticleData): 

458 self.process_resource_metadata(xarticle) 

459 for bibitem in xarticle.bibitems: 

460 bibitem.type = "unknown" 

461 update_data_for_jats(xarticle, with_label=False) 

462 

463 return xarticle 

464 

465 def download_file(self, url: str, headers={}): 

466 """ 

467 Downloads a page and returns its content (decoded string). 

468 """ 

469 

470 for attempt in range(3): 

471 session: CachedSession = get_session() # type:ignore 

472 response = session.get( 

473 url, 

474 headers=headers, 

475 ) 

476 

477 content = self.decode_response(response) 

478 if content == "" or not content: 

479 self.logger.debug("Got empty content while fetching ! ") 

480 # 15 mins, 30 mins, 45 mins 

481 delay_minutes = attempt * 15 

482 self.logger.debug( 

483 f"Retrying in {delay_minutes}mins ({(datetime.now() + timedelta(minutes=delay_minutes)).time()})", 

484 extra={"url": url}, 

485 ) 

486 self.pause_function(delay_minutes * 60) 

487 continue 

488 return content 

489 raise ValueError(f"Could not decode content at {url}") 

490 

491 def decode_response(self, response: requests.Response, encoding: str | None = None): 

492 """Override this if the content-type headers from the sources are advertising something else than the actual content 

493 SASA needs this""" 

494 # Force 

495 if encoding: 

496 response.encoding = encoding 

497 return response.text 

498 

499 # Attempt to get encoding using HTTP headers 

500 content_type_tag = response.headers.get("Content-Type", None) 

501 

502 if content_type_tag: 502 ↛ 509line 502 didn't jump to line 509 because the condition on line 502 was always true

503 charset = self.parse_content_type_charset(content_type_tag) 

504 if charset: 504 ↛ 505line 504 didn't jump to line 505 because the condition on line 504 was never true

505 response.encoding = charset 

506 return response.text 

507 

508 # Attempt to get encoding using HTML meta charset tag 

509 soup = BeautifulSoup(response.text, "html5lib") 

510 charset = soup.select_one("meta[charset]") 

511 if charset: 

512 htmlencoding = charset.get("charset") 

513 if isinstance(htmlencoding, str): 513 ↛ 518line 513 didn't jump to line 518 because the condition on line 513 was always true

514 response.encoding = htmlencoding 

515 return response.text 

516 

517 # Attempt to get encoding using HTML meta content type tag 

518 content_type_tag = soup.select_one( 

519 'meta[http-equiv="Content-Type"],meta[http-equiv="content-type"]' 

520 ) 

521 if content_type_tag: 

522 content_type = content_type_tag.get("content") 

523 if isinstance(content_type, str): 523 ↛ 529line 523 didn't jump to line 529 because the condition on line 523 was always true

524 charset = self.parse_content_type_charset(content_type) 

525 if charset: 525 ↛ 529line 525 didn't jump to line 529 because the condition on line 525 was always true

526 response.encoding = charset 

527 return response.text 

528 

529 return response.text 

530 

531 @staticmethod 

532 def parse_content_type_charset(content_type: str): 

533 header = EmailPolicy.header_factory("content-type", content_type) 

534 if "charset" in header.params: 

535 return header.params.get("charset") 

536 

537 @tracer.start_as_current_span("add_xissue_to_database") 

538 def add_xissue_into_database(self, xissue: IssueData) -> IssueData: 

539 xissue.journal = self.collection 

540 xissue.source = self.source_domain 

541 

542 if xissue.fyear == 0: 

543 raise ValueError("Failsafe : Cannot insert issue without a year") 

544 

545 xpub = create_publisherdata() 

546 xpub.name = self.publisher 

547 xissue.publisher = xpub 

548 xissue.last_modified_iso_8601_date_str = timezone.now().isoformat() 

549 

550 attempt = 1 

551 success = False 

552 

553 while not success and attempt < 4: 

554 try: 

555 params = {"xissue": xissue, "use_body": False} 

556 cmd = addOrUpdateGDMLIssueXmlCmd(params) 

557 cmd.do() 

558 success = True 

559 self.logger.debug(f"Issue {xissue.pid} inserted in database") 

560 return xissue 

561 except SolrError: 

562 self.logger.warning( 

563 f"Encoutered SolrError while inserting issue {xissue.pid} in database" 

564 ) 

565 attempt += 1 

566 self.logger.debug(f"Attempt {attempt}. sleeping 10 seconds.") 

567 self.pause_function(10) 

568 except Exception as e: 

569 self.logger.error( 

570 f"Got exception while attempting to insert {xissue.pid} in database : {e}" 

571 ) 

572 raise e 

573 

574 if success is False: 

575 raise ConnectionRefusedError("Cannot connect to SolR") 

576 

577 assert False, "Unreachable" 

578 

579 def get_metadata_using_citation_meta( 

580 self, 

581 xarticle: ArticleData, 

582 xissue: IssueData, 

583 soup: BeautifulSoup, 

584 what: list[CitationLiteral] = [], 

585 ): 

586 """ 

587 :param xarticle: the xarticle that will collect the metadata 

588 :param xissue: the xissue that will collect the publisher 

589 :param soup: the BeautifulSoup object of tha article page 

590 :param what: list of citation_ items to collect. 

591 :return: None. The given article is modified 

592 """ 

593 

594 if "title" in what: 

595 # TITLE 

596 citation_title_node = soup.select_one("meta[name='citation_title']") 

597 if citation_title_node: 597 ↛ 602line 597 didn't jump to line 602 because the condition on line 597 was always true

598 title = citation_title_node.get("content") 

599 if isinstance(title, str): 599 ↛ 602line 599 didn't jump to line 602 because the condition on line 599 was always true

600 xarticle.title_tex = title 

601 

602 if "author" in what: 602 ↛ 631line 602 didn't jump to line 631 because the condition on line 602 was always true

603 # AUTHORS 

604 citation_author_nodes = soup.select("meta[name^='citation_author']") 

605 current_author: ContributorDict | None = None 

606 for citation_author_node in citation_author_nodes: 

607 if citation_author_node.get("name") == "citation_author": 

608 text_author = citation_author_node.get("content") 

609 if not isinstance(text_author, str): 609 ↛ 610line 609 didn't jump to line 610 because the condition on line 609 was never true

610 raise ValueError("Cannot parse author") 

611 if text_author == "": 611 ↛ 612line 611 didn't jump to line 612 because the condition on line 611 was never true

612 current_author = None 

613 continue 

614 current_author = create_contributor(role="author", string_name=text_author) 

615 xarticle.contributors.append(current_author) 

616 continue 

617 if current_author is None: 617 ↛ 618line 617 didn't jump to line 618 because the condition on line 617 was never true

618 self.logger.warning("Couldn't parse citation author") 

619 continue 

620 if citation_author_node.get("name") == "citation_author_institution": 

621 text_institution = citation_author_node.get("content") 

622 if not isinstance(text_institution, str): 622 ↛ 623line 622 didn't jump to line 623 because the condition on line 622 was never true

623 continue 

624 current_author["addresses"].append(text_institution) 

625 if citation_author_node.get("name") == "citation_author_ocrid": 625 ↛ 626line 625 didn't jump to line 626 because the condition on line 625 was never true

626 text_orcid = citation_author_node.get("content") 

627 if not isinstance(text_orcid, str): 

628 continue 

629 current_author["orcid"] = text_orcid 

630 

631 if "pdf" in what: 

632 # PDF 

633 citation_pdf_node = soup.select_one('meta[name="citation_pdf_url"]') 

634 if citation_pdf_node: 

635 pdf_url = citation_pdf_node.get("content") 

636 if isinstance(pdf_url, str): 636 ↛ 639line 636 didn't jump to line 639 because the condition on line 636 was always true

637 add_pdf_link_to_xarticle(xarticle, pdf_url) 

638 

639 if "lang" in what: 

640 # LANG 

641 citation_lang_node = soup.select_one("meta[name='citation_language']") 

642 if citation_lang_node: 642 ↛ 648line 642 didn't jump to line 648 because the condition on line 642 was always true

643 # TODO: check other language code 

644 content_text = citation_lang_node.get("content") 

645 if isinstance(content_text, str): 645 ↛ 648line 645 didn't jump to line 648 because the condition on line 645 was always true

646 xarticle.lang = standardize_tag(content_text) 

647 

648 if "abstract" in what: 

649 # ABSTRACT 

650 abstract_node = soup.select_one("meta[name='citation_abstract']") 

651 if abstract_node is not None: 

652 abstract = abstract_node.get("content") 

653 if not isinstance(abstract, str): 653 ↛ 654line 653 didn't jump to line 654 because the condition on line 653 was never true

654 raise ValueError("Couldn't parse abstract from meta") 

655 abstract = BeautifulSoup(abstract, "html.parser").text 

656 lang = abstract_node.get("lang") 

657 if not isinstance(lang, str): 

658 lang = self.detect_language(abstract, xarticle) 

659 xarticle.abstracts.append(create_abstract(lang=lang, value_tex=abstract)) 

660 

661 if "page" in what: 

662 # PAGES 

663 citation_fpage_node = soup.select_one("meta[name='citation_firstpage']") 

664 if citation_fpage_node: 

665 page = citation_fpage_node.get("content") 

666 if isinstance(page, str): 666 ↛ 671line 666 didn't jump to line 671 because the condition on line 666 was always true

667 page = page.split("(")[0] 

668 if len(page) < 32: 668 ↛ 671line 668 didn't jump to line 671 because the condition on line 668 was always true

669 xarticle.fpage = page 

670 

671 citation_lpage_node = soup.select_one("meta[name='citation_lastpage']") 

672 if citation_lpage_node: 

673 page = citation_lpage_node.get("content") 

674 if isinstance(page, str): 674 ↛ 679line 674 didn't jump to line 679 because the condition on line 674 was always true

675 page = page.split("(")[0] 

676 if len(page) < 32: 676 ↛ 679line 676 didn't jump to line 679 because the condition on line 676 was always true

677 xarticle.lpage = page 

678 

679 if "doi" in what: 

680 # DOI 

681 citation_doi_node = soup.select_one("meta[name='citation_doi']") 

682 if citation_doi_node: 682 ↛ 691line 682 didn't jump to line 691 because the condition on line 682 was always true

683 doi = citation_doi_node.get("content") 

684 if isinstance(doi, str): 684 ↛ 691line 684 didn't jump to line 691 because the condition on line 684 was always true

685 doi = doi.strip() 

686 pos = doi.find("10.") 

687 if pos > 0: 687 ↛ 688line 687 didn't jump to line 688 because the condition on line 687 was never true

688 doi = doi[pos:] 

689 xarticle.doi = doi 

690 

691 if "mr" in what: 

692 # MR 

693 citation_mr_node = soup.select_one("meta[name='citation_mr']") 

694 if citation_mr_node: 694 ↛ 703line 694 didn't jump to line 703 because the condition on line 694 was always true

695 mr = citation_mr_node.get("content") 

696 if isinstance(mr, str): 696 ↛ 703line 696 didn't jump to line 703 because the condition on line 696 was always true

697 mr = mr.strip() 

698 if mr.find("MR") == 0: 698 ↛ 703line 698 didn't jump to line 703 because the condition on line 698 was always true

699 mr = mr[2:] 

700 extid = create_extid("mr-item-id", mr) 

701 xarticle.extids.append(extid) 

702 

703 if "zbl" in what: 

704 # ZBL 

705 citation_zbl_node = soup.select_one("meta[name='citation_zbl']") 

706 if citation_zbl_node: 706 ↛ 715line 706 didn't jump to line 715 because the condition on line 706 was always true

707 zbl = citation_zbl_node.get("content") 

708 if isinstance(zbl, str): 708 ↛ 715line 708 didn't jump to line 715 because the condition on line 708 was always true

709 zbl = zbl.strip() 

710 if zbl.find("Zbl") == 0: 710 ↛ 715line 710 didn't jump to line 715 because the condition on line 710 was always true

711 zbl = zbl[3:].strip() 

712 extid = create_extid("zbl-item-id", zbl) 

713 xarticle.extids.append(extid) 

714 

715 if "publisher" in what: 

716 # PUBLISHER 

717 citation_publisher_node = soup.select_one("meta[name='citation_publisher']") 

718 if citation_publisher_node: 718 ↛ 727line 718 didn't jump to line 727 because the condition on line 718 was always true

719 pub = citation_publisher_node.get("content") 

720 if isinstance(pub, str): 720 ↛ 727line 720 didn't jump to line 727 because the condition on line 720 was always true

721 pub = pub.strip() 

722 if pub != "": 722 ↛ 727line 722 didn't jump to line 727 because the condition on line 722 was always true

723 xpub = create_publisherdata() 

724 xpub.name = pub 

725 xissue.publisher = xpub 

726 

727 if "keywords" in what: 

728 # KEYWORDS 

729 citation_kwd_nodes = soup.select("meta[name='citation_keywords']") 

730 for kwd_node in citation_kwd_nodes: 

731 kwds = kwd_node.get("content") 

732 if isinstance(kwds, str): 732 ↛ 730line 732 didn't jump to line 730 because the condition on line 732 was always true

733 kwds = kwds.split(",") 

734 for kwd in kwds: 

735 if kwd == "": 

736 continue 

737 kwd = kwd.strip() 

738 xarticle.kwds.append({"type": "", "lang": xarticle.lang, "value": kwd}) 

739 

740 if "references" in what: 

741 citation_references = soup.select("meta[name='citation_reference']") 

742 for index, tag in enumerate(citation_references): 

743 content = tag.get("content") 

744 if not isinstance(content, str): 744 ↛ 745line 744 didn't jump to line 745 because the condition on line 744 was never true

745 raise ValueError("Cannot parse citation_reference meta") 

746 label = str(index + 1) 

747 if regex.match(r"^\[\d+\].*", content): 747 ↛ 748line 747 didn't jump to line 748 because the condition on line 747 was never true

748 label = None 

749 xarticle.bibitems.append(self.__parse_meta_citation_reference(content, label)) 

750 

751 def get_metadata_using_dcterms( 

752 self, 

753 xarticle: ArticleData, 

754 soup: "Tag", 

755 what: "Iterable[Literal['abstract', 'keywords', 'date_published', 'article_type']]", 

756 ): 

757 if "abstract" in what: 757 ↛ 765line 757 didn't jump to line 765 because the condition on line 757 was always true

758 abstract_tag = soup.select_one("meta[name='DCTERMS.abstract']") 

759 if abstract_tag: 759 ↛ 765line 759 didn't jump to line 765 because the condition on line 759 was always true

760 abstract_text = self.get_str_attr(abstract_tag, "content") 

761 xarticle.abstracts.append( 

762 create_abstract(lang="en", value_tex=cleanup_str(abstract_text)) 

763 ) 

764 

765 if "keywords" in what: 765 ↛ 774line 765 didn't jump to line 774 because the condition on line 765 was always true

766 keyword_tags = soup.select("meta[name='DC.subject']") 

767 for tag in keyword_tags: 

768 kwd_text = tag.get("content") 

769 if not isinstance(kwd_text, str) or len(kwd_text) == 0: 769 ↛ 770line 769 didn't jump to line 770 because the condition on line 769 was never true

770 continue 

771 kwd = create_subj(value=kwd_text) 

772 xarticle.kwds.append(kwd) 

773 

774 if "date_published" in what: 774 ↛ 775line 774 didn't jump to line 775 because the condition on line 774 was never true

775 published_tag = soup.select_one("meta[name='DC.Date.created']") 

776 if published_tag: 

777 published_text = self.get_str_attr(published_tag, "content") 

778 xarticle.date_published = published_text 

779 

780 if "article_type" in what: 780 ↛ 781line 780 didn't jump to line 781 because the condition on line 780 was never true

781 type_tag = soup.select_one("meta[name='DC.Type.articleType']") 

782 if type_tag: 

783 type_text = self.get_str_attr(type_tag, "content") 

784 xarticle.atype = type_text 

785 

786 def create_xissue( 

787 self, 

788 url: str | None, 

789 year: int, 

790 volume_number: str | None, 

791 issue_number: str | None = None, 

792 vseries: str | None = None, 

793 lyear: int | None = None, 

794 ): 

795 if url is not None and url.endswith("/"): 795 ↛ 796line 795 didn't jump to line 796 because the condition on line 795 was never true

796 url = url[:-1] 

797 xissue = create_issuedata() 

798 xissue.url = url 

799 

800 year_str = year 

801 xissue.fyear = year 

802 if lyear: 802 ↛ 803line 802 didn't jump to line 803 because the condition on line 802 was never true

803 year_str = f"{year}_{lyear}" 

804 xissue.lyear = lyear 

805 xissue.pid = self.get_issue_pid( 

806 self.collection_id, year_str, volume_number, issue_number, vseries 

807 ) 

808 

809 if volume_number is not None: 

810 xissue.volume = regex.sub(r"[^\w-]+", "_", volume_number) 

811 

812 if issue_number is not None: 

813 xissue.number = issue_number.replace(",", "-") 

814 

815 if vseries is not None: 815 ↛ 816line 815 didn't jump to line 816 because the condition on line 815 was never true

816 xissue.vseries = vseries 

817 return xissue 

818 

819 def detect_language(self, text: str, article: ArticleData | None = None): 

820 if article and article.lang is not None and article.lang != "und": 

821 return article.lang 

822 

823 language = self.language_detector.detect_language_of(text) 

824 

825 if not language: 825 ↛ 826line 825 didn't jump to line 826 because the condition on line 825 was never true

826 return "und" 

827 return language.iso_code_639_1.name.lower() 

828 

829 def get_str_attr(self, tag: "Tag", attr: str): 

830 """Equivalent of `tag.get(attr)`, but ensures the return value is a string""" 

831 node_attr = tag.get(attr) 

832 if isinstance(node_attr, list): 832 ↛ 833line 832 didn't jump to line 833 because the condition on line 832 was never true

833 raise ValueError( 

834 f"[{self.source_domain}] {self.collection_id} : html tag has multiple {attr} attributes." 

835 ) 

836 if node_attr is None: 836 ↛ 837line 836 didn't jump to line 837 because the condition on line 836 was never true

837 raise ValueError( 

838 f"[{self.source_domain}] {self.collection_id} : html tag doesn't have any {attr} attributes" 

839 ) 

840 return node_attr 

841 

842 def create_trans_title( 

843 self, 

844 resource_type: str, 

845 title_str: str, 

846 lang: str, 

847 xresource_lang: str, 

848 title_type: str = "main", 

849 ): 

850 tag = "trans-title" if resource_type == "article" else "issue-title" 

851 

852 ckeditor_data = build_jats_data_from_html_field( 

853 title_str, 

854 tag=tag, 

855 text_lang=lang, 

856 resource_lang=xresource_lang, 

857 delimiter_inline=self.delimiter_inline_formula, 

858 delimiter_disp=self.delimiter_disp_formula, 

859 ) 

860 

861 titledata = create_titledata( 

862 lang=lang, 

863 type="main", 

864 title_html=ckeditor_data["value_html"], 

865 title_xml=ckeditor_data["value_xml"], 

866 ) 

867 

868 return titledata 

869 

870 references_mapping = { 

871 "citation_title": get_article_title_xml, 

872 "citation_journal_title": get_source_xml, 

873 "citation_publication_date": get_year_xml, 

874 "citation_firstpage": get_fpage_xml, 

875 "citation_lastpage": get_lpage_xml, 

876 } 

877 

878 @classmethod 

879 def __parse_meta_citation_reference(cls, content: str, label=None): 

880 categories = content.split(";") 

881 

882 if len(categories) == 1: 

883 return parse_mixed_citation_into_ref(content, label=label) 

884 

885 citation_data = [c.split("=") for c in categories if "=" in c] 

886 del categories 

887 

888 xml_string = "" 

889 authors_parsed = False 

890 authors_strings = [] 

891 for data in citation_data: 

892 key = data[0].strip() 

893 citation_content = data[1] 

894 if key == "citation_author": 

895 authors_strings.append(get_author_xml(template_str=citation_content)) 

896 continue 

897 elif not authors_parsed: 

898 xml_string += ", ".join(authors_strings) 

899 authors_parsed = True 

900 

901 if key in cls.references_mapping: 

902 xml_string += " " + cls.references_mapping[key](citation_content) 

903 

904 return parse_mixed_citation_into_ref(xml_string, label=label) 

905 

906 @classmethod 

907 def get_or_create_source(cls): 

908 source, created = Source.objects.get_or_create( 

909 domain=cls.source_domain, 

910 defaults={ 

911 "name": cls.source_name, 

912 "website": cls.source_website, 

913 "view_id": cls.get_view_id(), 

914 }, 

915 ) 

916 if created: 

917 source.save() 

918 return source 

919 

920 @staticmethod 

921 def get_issue_pid( 

922 collection_id: str, 

923 year: int | str, 

924 volume_number: str | None = None, 

925 issue_number: str | None = None, 

926 series: str | None = None, 

927 ): 

928 # Replace any non-word character with an underscore 

929 pid = f"{collection_id}_{year}" 

930 if series is not None: 930 ↛ 931line 930 didn't jump to line 931 because the condition on line 930 was never true

931 pid += f"_{series}" 

932 if volume_number is not None: 

933 pid += f"_{volume_number}" 

934 if issue_number is not None: 

935 pid += f"_{issue_number}" 

936 pid = regex.sub(r"[^\w-]+", "_", cleanup_str(pid)) 

937 return pid 

938 

939 @staticmethod 

940 def set_pages(article: ArticleData, pages: str, separator: str = "-"): 

941 pages_split = pages.split(separator) 

942 if len(pages_split) == 0: 942 ↛ 943line 942 didn't jump to line 943 because the condition on line 942 was never true

943 article.page_range = pages 

944 if len(pages_split) > 0: 944 ↛ exitline 944 didn't return from function 'set_pages' because the condition on line 944 was always true

945 if pages[0].isnumeric(): 945 ↛ exitline 945 didn't return from function 'set_pages' because the condition on line 945 was always true

946 article.fpage = pages_split[0] 

947 if ( 947 ↛ 952line 947 didn't jump to line 952 because the condition on line 947 was never true

948 len(pages_split) > 1 

949 and pages_split[0] != pages_split[1] 

950 and pages_split[1].isnumeric() 

951 ): 

952 article.lpage = pages_split[1] 

953 

954 @staticmethod 

955 def _process_pdf_header(chunk: str, response: requests.Response | aiohttp.ClientResponse): 

956 content_type = response.headers.get("Content-Type") 

957 if regex.match(rb"^%PDF-\d\.\d", chunk): 

958 if content_type and "application/pdf" in content_type: 

959 # The file is unmistakably a pdf 

960 return [ 

961 True, 

962 response, 

963 { 

964 "status": ExtlinkChecked.Status.OK, 

965 "message": "", 

966 }, 

967 ] 

968 # The file is a pdf, but the content type advertised by the server is wrong 

969 return [ 

970 True, 

971 response, 

972 { 

973 "status": ExtlinkChecked.Status.WARNING, 

974 "message": f"Content-Type header: {content_type}", 

975 }, 

976 ] 

977 

978 # Reaching here means we couldn't find the pdf. 

979 if not content_type or "application/pdf" not in content_type: 

980 return [ 

981 False, 

982 response, 

983 { 

984 "status": ExtlinkChecked.Status.ERROR, 

985 "message": f"Content-Type header: {content_type}; PDF Header not found: got {chunk}", 

986 }, 

987 ] 

988 

989 return [ 

990 False, 

991 response, 

992 { 

993 "status": ExtlinkChecked.Status.ERROR, 

994 "message": f"PDF Header not found: got {chunk}", 

995 }, 

996 ] 

997 

998 @classmethod 

999 async def a_check_pdf_link_validity( 

1000 cls, url: str, verify=True 

1001 ) -> list[bool | aiohttp.ClientResponse | dict]: 

1002 """ 

1003 Check the validity of the PDF links. 

1004 """ 

1005 CHUNK_SIZE = 10 # Nombre de caractères à récupérer 

1006 headers = { 

1007 "Range": f"bytes=0-{CHUNK_SIZE}", 

1008 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0", 

1009 } 

1010 async with cls.async_session.get( 

1011 url, headers=headers, allow_redirects=True, ssl=verify 

1012 ) as response: 

1013 try: 

1014 chunk = await response.content.read(CHUNK_SIZE) 

1015 return BaseCollectionCrawler._process_pdf_header(chunk, response) 

1016 except StopIteration: 

1017 return [ 

1018 False, 

1019 response, 

1020 { 

1021 "status": ExtlinkChecked.Status.ERROR, 

1022 "message": "Error reading PDF header", 

1023 }, 

1024 ] 

1025 

1026 @classmethod 

1027 def check_pdf_link_validity( 

1028 cls, url: str, verify=True 

1029 ) -> list[bool | requests.Response | None | dict]: 

1030 """ 

1031 Check the validity of the PDF links. 

1032 """ 

1033 CHUNK_SIZE = 10 # Nombre de caractères à récupérer 

1034 header = { 

1035 "Range": f"bytes=0-{CHUNK_SIZE}", 

1036 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0", 

1037 } 

1038 with get_session().get( 

1039 url, headers=header, allow_redirects=True, verify=verify, stream=True 

1040 ) as response: 

1041 try: 

1042 chunk = next(response.iter_content(CHUNK_SIZE)) 

1043 return BaseCollectionCrawler._process_pdf_header(chunk, response) 

1044 except StopIteration: 

1045 return [ 

1046 False, 

1047 response, 

1048 { 

1049 "status": ExtlinkChecked.Status.ERROR, 

1050 "message": "Error reading PDF header", 

1051 }, 

1052 ] 

1053 

1054 @classmethod 

1055 async def check_extlink_validity(cls, extlink: "ExtLink"): 

1056 """ 

1057 Method used by rot_monitoring to check if links have expired 

1058 """ 

1059 defaults: dict = {"date": datetime.now(), "status": ExtlinkChecked.Status.OK} 

1060 header = { 

1061 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0" 

1062 } 

1063 verify = True 

1064 if not cls.verify: 

1065 verify = False 

1066 try: 

1067 # For the GDZ links, we just check if the http response is 200 or 206 

1068 if ( 

1069 extlink.rel == "article-pdf" 

1070 and "gdz.sub.uni-goettingen.de" not in extlink.location 

1071 ): 

1072 isok, response, message = await cls.a_check_pdf_link_validity( 

1073 extlink.location, verify 

1074 ) 

1075 defaults.update(message) 

1076 defaults["http_status"] = response.status 

1077 else: 

1078 async with cls.async_session.get( 

1079 url=extlink.location, 

1080 headers=header, 

1081 allow_redirects=True, 

1082 ssl=verify, 

1083 ) as response: 

1084 defaults["http_status"] = response.status 

1085 if response.status not in (200, 206): 

1086 defaults["status"] = ExtlinkChecked.Status.ERROR 

1087 

1088 except aiohttp.ClientSSLError: 

1089 cls.logger.error("SSL error for the url: %s", extlink.location) 

1090 defaults["status"] = ExtlinkChecked.Status.ERROR 

1091 defaults["message"] = "SSL error" 

1092 except aiohttp.ClientConnectionError: 

1093 cls.logger.error("Connection error for the url: %s", extlink.location) 

1094 defaults["status"] = ExtlinkChecked.Status.ERROR 

1095 defaults["message"] = "Connection error" 

1096 except TimeoutError: 

1097 cls.logger.error("Timeout error for the url: %s", extlink.location) 

1098 defaults["status"] = ExtlinkChecked.Status.ERROR 

1099 defaults["message"] = "Timeout error" 

1100 finally: 

1101 try: 

1102 await ExtlinkChecked.objects.aupdate_or_create(extlink=extlink, defaults=defaults) 

1103 cls.logger.info( 

1104 "DB Update, source: %s, url: %s", cls.source_domain, extlink.location 

1105 ) 

1106 except IntegrityError: 

1107 cls.logger.error( 

1108 "Extlink was deleted, source: %s, url: %s", cls.source_domain, extlink.location 

1109 ) 

1110 

1111 def resolve_year_end(self, pid: str, default: int) -> int: 

1112 if pid in self.pid_year_restrictions: 

1113 return datetime.now().year - self.pid_year_restrictions[pid] 

1114 return default 

1115 

1116 

1117def get_first_last_years(years_str: str): 

1118 years_splitted = years_str.split("-") 

1119 fyear = int(years_splitted[0]) 

1120 lyear = None 

1121 if len(years_splitted) > 1: 

1122 lyear = int(years_splitted[1]) 

1123 return fyear, lyear