Coverage for src/crawler/abstract_crawlers/base_crawler.py: 63%
610 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import logging
2import time
3from collections.abc import Iterable
4from datetime import datetime, timedelta
5from email.policy import EmailPolicy
6from typing import TYPE_CHECKING, Literal
8import aiohttp
9import regex
10import requests
11from bs4 import BeautifulSoup
12from django.conf import settings
13from django.contrib.auth.models import User
14from django.db.utils import IntegrityError
15from django.utils import timezone
16from langcodes import standardize_tag
17from lingua import LanguageDetector, LanguageDetectorBuilder
18from opentelemetry import trace
19from ptf.cmds.xml.ckeditor.utils import (
20 build_jats_data_from_html_field,
21)
22from ptf.cmds.xml.jats.builder.references import (
23 get_article_title_xml,
24 get_author_xml,
25 get_fpage_xml,
26 get_lpage_xml,
27 get_source_xml,
28 get_year_xml,
29)
30from ptf.cmds.xml.jats.jats_parser import parse_mixed_citation_into_ref
31from ptf.external.session import get_session
32from ptf.model_data import (
33 ArticleData,
34 ContributorDict,
35 IssueData,
36 ResourceData,
37 TitleDict,
38 create_abstract,
39 create_contributor,
40 create_extid,
41 create_issuedata,
42 create_publisherdata,
43 create_subj,
44 create_titledata,
45)
46from ptf.model_data_converter import update_data_for_jats
47from ptf.models import ExtLink
48from pylatexenc.latex2text import LatexNodes2Text
49from pysolr import SolrError
50from requests_cache import CachedSession
52from crawler.cmds.xml_cmds import addOrUpdateGDMLIssueXmlCmd
53from crawler.models import Source
54from crawler.models.extlink_checked import ExtlinkChecked
55from crawler.types import CitationLiteral
56from crawler.utils import (
57 add_pdf_link_to_xarticle,
58 add_source_link_to_xarticle,
59 cleanup_str,
60 get_all_cols,
61 get_or_create_collection,
62)
64if TYPE_CHECKING:
65 from collections.abc import Callable
67 from bs4 import Tag
70class CrawlerTitleDict(TitleDict):
71 title_tex: str | None
74class BaseCollectionCrawler:
75 """
76 Base collection for the crawlers.
77 To create a crawler:
78 1) derive a class from BaseCollectionCrawler and name it XXXCrawler
79 2) override the functions parse_collection_content, parse_issue_content and parse_article_content
80 3) update factory.py so that crawler_factory can return your new crawler
81 """
83 logger = logging.getLogger(__name__)
84 tracer = trace.get_tracer(__name__)
86 source_name = ""
87 source_domain = ""
88 source_website = ""
90 issue_href = ""
92 collection = None
93 source = None
94 user = None
95 session: requests.Session | CachedSession
96 async_session: aiohttp.ClientSession
97 is_checkable = True
98 verify = True
99 headers = {}
101 requests_interval = getattr(settings, "REQUESTS_INTERVAL", 90)
102 "seconds to wait between two http requests"
103 requests_timeout = 60
104 "seconds to wait before aborting the connection (if no bytes are recieved)"
106 latext_parser = LatexNodes2Text()
108 # Override the values in your concrete crawler if the formulas in text (titles, abstracts)
109 # do not use the "$" to surround tex formulas
110 delimiter_inline_formula = "$"
111 delimiter_disp_formula = "$"
113 # HACK : Workaround for tests (monkeypatching)
114 # We store the class here, so we can monkeypatch it when running tests
115 # subCrawlers = {
116 # LofplCrawler: None
117 # }
118 subCrawlers: dict[type["BaseCollectionCrawler"], "BaseCollectionCrawler | None"] = {}
120 _language_detector: LanguageDetector | None = None
121 _language_detector_builder = LanguageDetectorBuilder.from_all_languages()
123 match_headers = False
124 "Whereas to include headers in requests cache key"
125 orcid_re = r"https\:\/\/orcid\.org\/(?P<orcid>\d{4}-\d{4}-\d{4}-\d{4})"
127 ignore_missing_pdf = True
128 "Set this to False on a Crawler-basis to allow inserting articles without PDFs"
129 pid_year_restrictions: dict[str, int] = {}
130 "pid -> excluded years count"
132 pause_function: "Callable[[int], None]"
133 "Overridable the pause function (used in celery tasks to speedup aborting)"
135 @classmethod
136 def get_view_id(cls):
137 return cls.source_domain
139 @property
140 def language_detector(self):
141 """Crawler Instance singleton for language builder.
142 Late init of LanguageDetector to save on memory"""
143 if not self._language_detector:
144 self._language_detector = self._language_detector_builder.build()
145 return self._language_detector
147 def __init__(
148 self,
149 *args,
150 username: str,
151 collection_id: str,
152 dry: bool = False,
153 publisher: str = "",
154 collection_url: str | None = None,
155 backend=None,
156 pause_function=staticmethod(time.sleep),
157 ):
158 if not collection_url: 158 ↛ 159line 158 didn't jump to line 159 because the condition on line 158 was never true
159 all_cols = get_all_cols()
160 col = all_cols[collection_id]
162 collection_url = col["sources"].get(self.source_domain, None)
163 if collection_url is None:
164 raise ValueError(
165 f"Source {self.source_domain} not found for collection {collection_id}"
166 )
167 self.collection_url = collection_url
168 for CrawlerClass in self.subCrawlers: 168 ↛ 169line 168 didn't jump to line 169 because the loop on line 168 never started
169 self.subCrawlers[CrawlerClass] = CrawlerClass(
170 *args,
171 username=username,
172 collection_id=collection_id,
173 dry=dry,
174 publisher=publisher,
175 collection_url=collection_url,
176 )
177 self.logger = logging.getLogger(__name__ + "." + self.source_domain)
178 # self.logger = logging.getLogger(__name__)
180 self.username = username
182 self.collection_id = collection_id
184 self.dry = dry
185 self.publisher = publisher
187 # Classproperty : We sometimes want to use the session without initializing the class (rot monitoring)
188 BaseCollectionCrawler.session = requests.Session()
190 self.pause_function = pause_function
192 # Skipped when running tests
193 self.initialize()
195 self.backend = backend
197 def initialize(self):
198 """
199 Acts as a "second" init function to skip model accesses during test data generation
200 """
201 self.collection = get_or_create_collection(self.collection_id)
202 self.source = self.get_or_create_source()
203 self.user = User.objects.get(username=self.username)
204 BaseCollectionCrawler.session = get_session()
205 BaseCollectionCrawler.session.verify = self.verify
206 self.session.pause_function = self.pause_function
207 self.session.delay = self.requests_interval
209 @classmethod
210 def can_crawl(cls, pid: str) -> bool:
211 return True
213 def parse_collection_content(self, content: str) -> list[IssueData]:
214 """
215 Parse the HTML content with BeautifulSoup
216 returns a list of xissue.
217 Override this function in a derived class
218 """
219 return []
221 def parse_issue_content(self, content: str, xissue: IssueData):
222 """
223 Parse the HTML content with BeautifulSoup
224 Fills the xissue.articles
225 Override this function in a derived class.
227 CAV : You are supposed to create articles there. Please assign a PID to each article.
228 The PID can be `a + article_index`, like this : `a0` `a21`
229 """
231 def parse_article_content(
232 self, content: str, xissue: IssueData, xarticle: ArticleData, url: str
233 ) -> ArticleData | None:
234 """
235 Parse the HTML content with BeautifulSoup
236 returns the xarticle.
237 Override this function in a derived class.
238 The xissue is passed to the function in case the article page has issue information (ex: publisher)
239 The article url is also passed as a parameter
241 CAV : You are supposed to assign articles pid again here
242 """
243 return xarticle
245 @tracer.start_as_current_span("crawl_collection")
246 def crawl_collection(self):
247 # TODO: Comments, filter
248 """
249 Crawl an entire collection. ptf.models.Container objects are created.
250 - get the HTML content of the collection_url
251 - parse the HTML content with beautifulsoup to extract the list of issues
252 - merge the xissues (some Source can have multiple pages for 1 volume/issue. We create only 1 container)
253 - crawl each issue if col_only is False
254 - Returns the list of merged issues.
255 It is an OrderedDict {pid: {"issues": xissues}}
256 The key is the pid of the merged issues.
257 Ex: The source may have Ex: Volume 6 (2000) and Volume 6 (1999)
258 the pid is then made with 1999-2000__6_
259 """
261 if self.source is None:
262 raise RuntimeError("ERROR: the source is not set")
264 content = self.download_file(self.collection_url)
265 if content:
266 xissues = self.parse_collection_content(content)
267 else:
268 # download_file returns None (404)
269 return None
271 """
272 Some collections split the same volumes in different pages
273 Ex: Volume 6 (2000) and Volume 6 (1999)
274 We merge the 2 xissues with the same volume number => Volume 6 (1999-2000)
275 """
276 # merged_xissues = self.merge_xissues(xissues)
278 xissues_dict = {str(i.pid): i for i in xissues}
280 return xissues_dict
282 def start_process_issue(self, xissue: IssueData):
283 # Some source, like EuDML do not have a separate HTML pages for an issue's table of content.
284 # The list of articles directly come from the collection HTML page: the xissue has no url attribute
285 issue_url = xissue.url
286 if issue_url is not None:
287 if issue_url.endswith(".pdf"):
288 add_pdf_link_to_xarticle(xissue, issue_url)
289 xissue.url = None
290 else:
291 content = self.download_file(issue_url)
292 with self.tracer.start_as_current_span("parse_issue_content"):
293 self.parse_issue_content(content, xissue)
295 @tracer.start_as_current_span("crawl_issue")
296 def crawl_issue(self, xissue: IssueData):
297 """
298 Crawl 1 wag page of an issue.
299 - get the HTML content of the issue
300 - parse the HTML content with beautifulsoup to extract the list of articles and/or the issue metadata
301 - crawl each article
302 """
304 self.start_process_issue(xissue)
306 xarticles = xissue.articles
308 parsed_xarticles = []
310 for xarticle in xarticles:
311 parsed_xarticle = self.crawl_article(xarticle, xissue)
312 if parsed_xarticle is not None:
313 parsed_xarticles.append(parsed_xarticle)
315 xissue.articles = parsed_xarticles
317 issue_has_pdf = self.article_has_pdf(xissue)
319 if self.ignore_missing_pdf:
320 xissue.articles = [a for a in xissue.articles if self.article_has_pdf(a)]
321 if self.dry:
322 return
323 if len(xissue.articles) == 0 and not issue_has_pdf:
324 return
325 self.process_resource_metadata(xissue, resource_type="issue")
327 self.add_xissue_into_database(xissue)
329 @staticmethod
330 def article_has_source(art: ArticleData | IssueData):
331 return (
332 next(
333 (e_link for e_link in art.ext_links if e_link["rel"] == "source"),
334 None,
335 )
336 is not None
337 )
339 @staticmethod
340 def article_has_pdf(art: ArticleData | IssueData):
341 return (
342 next(
343 (link for link in art.ext_links if link["rel"] in ["article-pdf", "article-html"]),
344 None,
345 )
346 is not None
347 )
349 def crawl_article(self, xarticle: ArticleData, xissue: IssueData):
350 # ARTICLE URL as en ExtLink (to display the link in the article page)
351 if xarticle.url is None:
352 if not self.article_has_source(xarticle): 352 ↛ 358line 352 didn't jump to line 358 because the condition on line 352 was always true
353 if xissue.url:
354 article_source = xissue.url
355 else:
356 article_source = self.collection_url
357 add_source_link_to_xarticle(xarticle, article_source, self.source_domain)
358 return self.process_article_metadata(xarticle)
360 if self.parse_article_content.__func__ != BaseCollectionCrawler.parse_article_content:
361 content = self.download_file(xarticle.url)
362 xarticle.pid = f"{xissue.pid}_{xarticle.pid}"
364 try:
365 with self.tracer.start_as_current_span("parse_article_content"):
366 parsed_xarticle = self.parse_article_content(
367 content, xissue, xarticle, xarticle.url
368 )
369 except ValueError as e:
370 self.logger.warning(e)
371 self.logger.warning("Retrying in 5 mins while invalidating cache")
372 self.pause_function(5 * 60)
373 content = self.download_file(xarticle.url, headers={"Cache-Control": "no-cache"})
374 with self.tracer.start_as_current_span("parse_article_content"):
375 parsed_xarticle = self.parse_article_content(
376 content, xissue, xarticle, xarticle.url
377 )
379 if not parsed_xarticle: 379 ↛ 380line 379 didn't jump to line 380 because the condition on line 379 was never true
380 return None
382 xarticle = parsed_xarticle
384 if xarticle.doi:
385 xarticle.pid = xarticle.doi.replace("/", "_").replace(".", "_").replace("-", "_")
387 if not self.article_has_source(xarticle) and xarticle.url:
388 add_source_link_to_xarticle(xarticle, xarticle.url, self.source_domain)
390 # The article title may have formulas surrounded with '$'
391 return self.process_article_metadata(xarticle)
393 def process_resource_metadata(self, xresource: ResourceData, resource_type="article"):
394 tag = "article-title" if resource_type == "article" else "issue-title"
396 # Process title tex
397 ckeditor_data = build_jats_data_from_html_field(
398 xresource.title_tex,
399 tag=tag,
400 text_lang=xresource.lang,
401 delimiter_inline=self.delimiter_inline_formula,
402 delimiter_disp=self.delimiter_disp_formula,
403 )
405 xresource.title_html = ckeditor_data["value_html"]
406 # xresource.title_tex = ckeditor_data["value_tex"]
407 xresource.title_xml = ckeditor_data["value_xml"]
409 abstracts_to_parse = [
410 xabstract for xabstract in xresource.abstracts if xabstract["tag"] == "abstract"
411 ]
412 # abstract may have formulas surrounded with '$'
413 if len(abstracts_to_parse) > 0:
414 for xabstract in abstracts_to_parse:
415 ckeditor_data = build_jats_data_from_html_field(
416 xabstract["value_tex"],
417 tag="abstract",
418 text_lang=xabstract["lang"],
419 resource_lang=xresource.lang,
420 field_type="abstract",
421 delimiter_inline=self.delimiter_inline_formula,
422 delimiter_disp=self.delimiter_disp_formula,
423 )
425 xabstract["value_html"] = ckeditor_data["value_html"]
426 # xabstract["value_tex"] = ckeditor_data["value_tex"]
427 xabstract["value_xml"] = ckeditor_data["value_xml"]
429 # Remove useless titles (duplicates)
430 titles = []
431 for trans_title in xresource.titles:
432 if trans_title["title_html"] != xresource.title_html: 432 ↛ 431line 432 didn't jump to line 431 because the condition on line 432 was always true
433 # trans_title["title_html"] = self.latex_converter.latex_to_text(
434 # trans_title["title_html"]
435 # )
436 titles.append(trans_title)
438 xresource.titles = titles
440 # remove duplicated keywords and add "msc" type to msc keywords :
441 xresource_keywords = []
442 for resource_keyword_dict in xresource.kwds:
443 resource_keyword = resource_keyword_dict["value"]
444 if resource_keyword == "" or resource_keyword in [ 444 ↛ 447line 444 didn't jump to line 447 because the condition on line 444 was never true
445 xresource_keyword["value"] for xresource_keyword in xresource_keywords
446 ]:
447 continue
449 if any(char.isdigit() for char in resource_keyword):
450 resource_keyword_dict["type"] = "msc"
452 xresource_keywords.append(resource_keyword_dict)
453 xresource.kwds = xresource_keywords
455 return xresource
457 def process_article_metadata(self, xarticle: ArticleData):
458 self.process_resource_metadata(xarticle)
459 for bibitem in xarticle.bibitems:
460 bibitem.type = "unknown"
461 update_data_for_jats(xarticle, with_label=False)
463 return xarticle
465 def download_file(self, url: str, headers={}):
466 """
467 Downloads a page and returns its content (decoded string).
468 """
470 for attempt in range(3):
471 session: CachedSession = get_session() # type:ignore
472 response = session.get(
473 url,
474 headers=headers,
475 )
477 content = self.decode_response(response)
478 if content == "" or not content:
479 self.logger.debug("Got empty content while fetching ! ")
480 # 15 mins, 30 mins, 45 mins
481 delay_minutes = attempt * 15
482 self.logger.debug(
483 f"Retrying in {delay_minutes}mins ({(datetime.now() + timedelta(minutes=delay_minutes)).time()})",
484 extra={"url": url},
485 )
486 self.pause_function(delay_minutes * 60)
487 continue
488 return content
489 raise ValueError(f"Could not decode content at {url}")
491 def decode_response(self, response: requests.Response, encoding: str | None = None):
492 """Override this if the content-type headers from the sources are advertising something else than the actual content
493 SASA needs this"""
494 # Force
495 if encoding:
496 response.encoding = encoding
497 return response.text
499 # Attempt to get encoding using HTTP headers
500 content_type_tag = response.headers.get("Content-Type", None)
502 if content_type_tag: 502 ↛ 509line 502 didn't jump to line 509 because the condition on line 502 was always true
503 charset = self.parse_content_type_charset(content_type_tag)
504 if charset: 504 ↛ 505line 504 didn't jump to line 505 because the condition on line 504 was never true
505 response.encoding = charset
506 return response.text
508 # Attempt to get encoding using HTML meta charset tag
509 soup = BeautifulSoup(response.text, "html5lib")
510 charset = soup.select_one("meta[charset]")
511 if charset:
512 htmlencoding = charset.get("charset")
513 if isinstance(htmlencoding, str): 513 ↛ 518line 513 didn't jump to line 518 because the condition on line 513 was always true
514 response.encoding = htmlencoding
515 return response.text
517 # Attempt to get encoding using HTML meta content type tag
518 content_type_tag = soup.select_one(
519 'meta[http-equiv="Content-Type"],meta[http-equiv="content-type"]'
520 )
521 if content_type_tag:
522 content_type = content_type_tag.get("content")
523 if isinstance(content_type, str): 523 ↛ 529line 523 didn't jump to line 529 because the condition on line 523 was always true
524 charset = self.parse_content_type_charset(content_type)
525 if charset: 525 ↛ 529line 525 didn't jump to line 529 because the condition on line 525 was always true
526 response.encoding = charset
527 return response.text
529 return response.text
531 @staticmethod
532 def parse_content_type_charset(content_type: str):
533 header = EmailPolicy.header_factory("content-type", content_type)
534 if "charset" in header.params:
535 return header.params.get("charset")
537 @tracer.start_as_current_span("add_xissue_to_database")
538 def add_xissue_into_database(self, xissue: IssueData) -> IssueData:
539 xissue.journal = self.collection
540 xissue.source = self.source_domain
542 if xissue.fyear == 0:
543 raise ValueError("Failsafe : Cannot insert issue without a year")
545 xpub = create_publisherdata()
546 xpub.name = self.publisher
547 xissue.publisher = xpub
548 xissue.last_modified_iso_8601_date_str = timezone.now().isoformat()
550 attempt = 1
551 success = False
553 while not success and attempt < 4:
554 try:
555 params = {"xissue": xissue, "use_body": False}
556 cmd = addOrUpdateGDMLIssueXmlCmd(params)
557 cmd.do()
558 success = True
559 self.logger.debug(f"Issue {xissue.pid} inserted in database")
560 return xissue
561 except SolrError:
562 self.logger.warning(
563 f"Encoutered SolrError while inserting issue {xissue.pid} in database"
564 )
565 attempt += 1
566 self.logger.debug(f"Attempt {attempt}. sleeping 10 seconds.")
567 self.pause_function(10)
568 except Exception as e:
569 self.logger.error(
570 f"Got exception while attempting to insert {xissue.pid} in database : {e}"
571 )
572 raise e
574 if success is False:
575 raise ConnectionRefusedError("Cannot connect to SolR")
577 assert False, "Unreachable"
579 def get_metadata_using_citation_meta(
580 self,
581 xarticle: ArticleData,
582 xissue: IssueData,
583 soup: BeautifulSoup,
584 what: list[CitationLiteral] = [],
585 ):
586 """
587 :param xarticle: the xarticle that will collect the metadata
588 :param xissue: the xissue that will collect the publisher
589 :param soup: the BeautifulSoup object of tha article page
590 :param what: list of citation_ items to collect.
591 :return: None. The given article is modified
592 """
594 if "title" in what:
595 # TITLE
596 citation_title_node = soup.select_one("meta[name='citation_title']")
597 if citation_title_node: 597 ↛ 602line 597 didn't jump to line 602 because the condition on line 597 was always true
598 title = citation_title_node.get("content")
599 if isinstance(title, str): 599 ↛ 602line 599 didn't jump to line 602 because the condition on line 599 was always true
600 xarticle.title_tex = title
602 if "author" in what: 602 ↛ 631line 602 didn't jump to line 631 because the condition on line 602 was always true
603 # AUTHORS
604 citation_author_nodes = soup.select("meta[name^='citation_author']")
605 current_author: ContributorDict | None = None
606 for citation_author_node in citation_author_nodes:
607 if citation_author_node.get("name") == "citation_author":
608 text_author = citation_author_node.get("content")
609 if not isinstance(text_author, str): 609 ↛ 610line 609 didn't jump to line 610 because the condition on line 609 was never true
610 raise ValueError("Cannot parse author")
611 if text_author == "": 611 ↛ 612line 611 didn't jump to line 612 because the condition on line 611 was never true
612 current_author = None
613 continue
614 current_author = create_contributor(role="author", string_name=text_author)
615 xarticle.contributors.append(current_author)
616 continue
617 if current_author is None: 617 ↛ 618line 617 didn't jump to line 618 because the condition on line 617 was never true
618 self.logger.warning("Couldn't parse citation author")
619 continue
620 if citation_author_node.get("name") == "citation_author_institution":
621 text_institution = citation_author_node.get("content")
622 if not isinstance(text_institution, str): 622 ↛ 623line 622 didn't jump to line 623 because the condition on line 622 was never true
623 continue
624 current_author["addresses"].append(text_institution)
625 if citation_author_node.get("name") == "citation_author_ocrid": 625 ↛ 626line 625 didn't jump to line 626 because the condition on line 625 was never true
626 text_orcid = citation_author_node.get("content")
627 if not isinstance(text_orcid, str):
628 continue
629 current_author["orcid"] = text_orcid
631 if "pdf" in what:
632 # PDF
633 citation_pdf_node = soup.select_one('meta[name="citation_pdf_url"]')
634 if citation_pdf_node:
635 pdf_url = citation_pdf_node.get("content")
636 if isinstance(pdf_url, str): 636 ↛ 639line 636 didn't jump to line 639 because the condition on line 636 was always true
637 add_pdf_link_to_xarticle(xarticle, pdf_url)
639 if "lang" in what:
640 # LANG
641 citation_lang_node = soup.select_one("meta[name='citation_language']")
642 if citation_lang_node: 642 ↛ 648line 642 didn't jump to line 648 because the condition on line 642 was always true
643 # TODO: check other language code
644 content_text = citation_lang_node.get("content")
645 if isinstance(content_text, str): 645 ↛ 648line 645 didn't jump to line 648 because the condition on line 645 was always true
646 xarticle.lang = standardize_tag(content_text)
648 if "abstract" in what:
649 # ABSTRACT
650 abstract_node = soup.select_one("meta[name='citation_abstract']")
651 if abstract_node is not None:
652 abstract = abstract_node.get("content")
653 if not isinstance(abstract, str): 653 ↛ 654line 653 didn't jump to line 654 because the condition on line 653 was never true
654 raise ValueError("Couldn't parse abstract from meta")
655 abstract = BeautifulSoup(abstract, "html.parser").text
656 lang = abstract_node.get("lang")
657 if not isinstance(lang, str):
658 lang = self.detect_language(abstract, xarticle)
659 xarticle.abstracts.append(create_abstract(lang=lang, value_tex=abstract))
661 if "page" in what:
662 # PAGES
663 citation_fpage_node = soup.select_one("meta[name='citation_firstpage']")
664 if citation_fpage_node:
665 page = citation_fpage_node.get("content")
666 if isinstance(page, str): 666 ↛ 671line 666 didn't jump to line 671 because the condition on line 666 was always true
667 page = page.split("(")[0]
668 if len(page) < 32: 668 ↛ 671line 668 didn't jump to line 671 because the condition on line 668 was always true
669 xarticle.fpage = page
671 citation_lpage_node = soup.select_one("meta[name='citation_lastpage']")
672 if citation_lpage_node:
673 page = citation_lpage_node.get("content")
674 if isinstance(page, str): 674 ↛ 679line 674 didn't jump to line 679 because the condition on line 674 was always true
675 page = page.split("(")[0]
676 if len(page) < 32: 676 ↛ 679line 676 didn't jump to line 679 because the condition on line 676 was always true
677 xarticle.lpage = page
679 if "doi" in what:
680 # DOI
681 citation_doi_node = soup.select_one("meta[name='citation_doi']")
682 if citation_doi_node: 682 ↛ 691line 682 didn't jump to line 691 because the condition on line 682 was always true
683 doi = citation_doi_node.get("content")
684 if isinstance(doi, str): 684 ↛ 691line 684 didn't jump to line 691 because the condition on line 684 was always true
685 doi = doi.strip()
686 pos = doi.find("10.")
687 if pos > 0: 687 ↛ 688line 687 didn't jump to line 688 because the condition on line 687 was never true
688 doi = doi[pos:]
689 xarticle.doi = doi
691 if "mr" in what:
692 # MR
693 citation_mr_node = soup.select_one("meta[name='citation_mr']")
694 if citation_mr_node: 694 ↛ 703line 694 didn't jump to line 703 because the condition on line 694 was always true
695 mr = citation_mr_node.get("content")
696 if isinstance(mr, str): 696 ↛ 703line 696 didn't jump to line 703 because the condition on line 696 was always true
697 mr = mr.strip()
698 if mr.find("MR") == 0: 698 ↛ 703line 698 didn't jump to line 703 because the condition on line 698 was always true
699 mr = mr[2:]
700 extid = create_extid("mr-item-id", mr)
701 xarticle.extids.append(extid)
703 if "zbl" in what:
704 # ZBL
705 citation_zbl_node = soup.select_one("meta[name='citation_zbl']")
706 if citation_zbl_node: 706 ↛ 715line 706 didn't jump to line 715 because the condition on line 706 was always true
707 zbl = citation_zbl_node.get("content")
708 if isinstance(zbl, str): 708 ↛ 715line 708 didn't jump to line 715 because the condition on line 708 was always true
709 zbl = zbl.strip()
710 if zbl.find("Zbl") == 0: 710 ↛ 715line 710 didn't jump to line 715 because the condition on line 710 was always true
711 zbl = zbl[3:].strip()
712 extid = create_extid("zbl-item-id", zbl)
713 xarticle.extids.append(extid)
715 if "publisher" in what:
716 # PUBLISHER
717 citation_publisher_node = soup.select_one("meta[name='citation_publisher']")
718 if citation_publisher_node: 718 ↛ 727line 718 didn't jump to line 727 because the condition on line 718 was always true
719 pub = citation_publisher_node.get("content")
720 if isinstance(pub, str): 720 ↛ 727line 720 didn't jump to line 727 because the condition on line 720 was always true
721 pub = pub.strip()
722 if pub != "": 722 ↛ 727line 722 didn't jump to line 727 because the condition on line 722 was always true
723 xpub = create_publisherdata()
724 xpub.name = pub
725 xissue.publisher = xpub
727 if "keywords" in what:
728 # KEYWORDS
729 citation_kwd_nodes = soup.select("meta[name='citation_keywords']")
730 for kwd_node in citation_kwd_nodes:
731 kwds = kwd_node.get("content")
732 if isinstance(kwds, str): 732 ↛ 730line 732 didn't jump to line 730 because the condition on line 732 was always true
733 kwds = kwds.split(",")
734 for kwd in kwds:
735 if kwd == "":
736 continue
737 kwd = kwd.strip()
738 xarticle.kwds.append({"type": "", "lang": xarticle.lang, "value": kwd})
740 if "references" in what:
741 citation_references = soup.select("meta[name='citation_reference']")
742 for index, tag in enumerate(citation_references):
743 content = tag.get("content")
744 if not isinstance(content, str): 744 ↛ 745line 744 didn't jump to line 745 because the condition on line 744 was never true
745 raise ValueError("Cannot parse citation_reference meta")
746 label = str(index + 1)
747 if regex.match(r"^\[\d+\].*", content): 747 ↛ 748line 747 didn't jump to line 748 because the condition on line 747 was never true
748 label = None
749 xarticle.bibitems.append(self.__parse_meta_citation_reference(content, label))
751 def get_metadata_using_dcterms(
752 self,
753 xarticle: ArticleData,
754 soup: "Tag",
755 what: "Iterable[Literal['abstract', 'keywords', 'date_published', 'article_type']]",
756 ):
757 if "abstract" in what: 757 ↛ 765line 757 didn't jump to line 765 because the condition on line 757 was always true
758 abstract_tag = soup.select_one("meta[name='DCTERMS.abstract']")
759 if abstract_tag: 759 ↛ 765line 759 didn't jump to line 765 because the condition on line 759 was always true
760 abstract_text = self.get_str_attr(abstract_tag, "content")
761 xarticle.abstracts.append(
762 create_abstract(lang="en", value_tex=cleanup_str(abstract_text))
763 )
765 if "keywords" in what: 765 ↛ 774line 765 didn't jump to line 774 because the condition on line 765 was always true
766 keyword_tags = soup.select("meta[name='DC.subject']")
767 for tag in keyword_tags:
768 kwd_text = tag.get("content")
769 if not isinstance(kwd_text, str) or len(kwd_text) == 0: 769 ↛ 770line 769 didn't jump to line 770 because the condition on line 769 was never true
770 continue
771 kwd = create_subj(value=kwd_text)
772 xarticle.kwds.append(kwd)
774 if "date_published" in what: 774 ↛ 775line 774 didn't jump to line 775 because the condition on line 774 was never true
775 published_tag = soup.select_one("meta[name='DC.Date.created']")
776 if published_tag:
777 published_text = self.get_str_attr(published_tag, "content")
778 xarticle.date_published = published_text
780 if "article_type" in what: 780 ↛ 781line 780 didn't jump to line 781 because the condition on line 780 was never true
781 type_tag = soup.select_one("meta[name='DC.Type.articleType']")
782 if type_tag:
783 type_text = self.get_str_attr(type_tag, "content")
784 xarticle.atype = type_text
786 def create_xissue(
787 self,
788 url: str | None,
789 year: int,
790 volume_number: str | None,
791 issue_number: str | None = None,
792 vseries: str | None = None,
793 lyear: int | None = None,
794 ):
795 if url is not None and url.endswith("/"): 795 ↛ 796line 795 didn't jump to line 796 because the condition on line 795 was never true
796 url = url[:-1]
797 xissue = create_issuedata()
798 xissue.url = url
800 year_str = year
801 xissue.fyear = year
802 if lyear: 802 ↛ 803line 802 didn't jump to line 803 because the condition on line 802 was never true
803 year_str = f"{year}_{lyear}"
804 xissue.lyear = lyear
805 xissue.pid = self.get_issue_pid(
806 self.collection_id, year_str, volume_number, issue_number, vseries
807 )
809 if volume_number is not None:
810 xissue.volume = regex.sub(r"[^\w-]+", "_", volume_number)
812 if issue_number is not None:
813 xissue.number = issue_number.replace(",", "-")
815 if vseries is not None: 815 ↛ 816line 815 didn't jump to line 816 because the condition on line 815 was never true
816 xissue.vseries = vseries
817 return xissue
819 def detect_language(self, text: str, article: ArticleData | None = None):
820 if article and article.lang is not None and article.lang != "und":
821 return article.lang
823 language = self.language_detector.detect_language_of(text)
825 if not language: 825 ↛ 826line 825 didn't jump to line 826 because the condition on line 825 was never true
826 return "und"
827 return language.iso_code_639_1.name.lower()
829 def get_str_attr(self, tag: "Tag", attr: str):
830 """Equivalent of `tag.get(attr)`, but ensures the return value is a string"""
831 node_attr = tag.get(attr)
832 if isinstance(node_attr, list): 832 ↛ 833line 832 didn't jump to line 833 because the condition on line 832 was never true
833 raise ValueError(
834 f"[{self.source_domain}] {self.collection_id} : html tag has multiple {attr} attributes."
835 )
836 if node_attr is None: 836 ↛ 837line 836 didn't jump to line 837 because the condition on line 836 was never true
837 raise ValueError(
838 f"[{self.source_domain}] {self.collection_id} : html tag doesn't have any {attr} attributes"
839 )
840 return node_attr
842 def create_trans_title(
843 self,
844 resource_type: str,
845 title_str: str,
846 lang: str,
847 xresource_lang: str,
848 title_type: str = "main",
849 ):
850 tag = "trans-title" if resource_type == "article" else "issue-title"
852 ckeditor_data = build_jats_data_from_html_field(
853 title_str,
854 tag=tag,
855 text_lang=lang,
856 resource_lang=xresource_lang,
857 delimiter_inline=self.delimiter_inline_formula,
858 delimiter_disp=self.delimiter_disp_formula,
859 )
861 titledata = create_titledata(
862 lang=lang,
863 type="main",
864 title_html=ckeditor_data["value_html"],
865 title_xml=ckeditor_data["value_xml"],
866 )
868 return titledata
870 references_mapping = {
871 "citation_title": get_article_title_xml,
872 "citation_journal_title": get_source_xml,
873 "citation_publication_date": get_year_xml,
874 "citation_firstpage": get_fpage_xml,
875 "citation_lastpage": get_lpage_xml,
876 }
878 @classmethod
879 def __parse_meta_citation_reference(cls, content: str, label=None):
880 categories = content.split(";")
882 if len(categories) == 1:
883 return parse_mixed_citation_into_ref(content, label=label)
885 citation_data = [c.split("=") for c in categories if "=" in c]
886 del categories
888 xml_string = ""
889 authors_parsed = False
890 authors_strings = []
891 for data in citation_data:
892 key = data[0].strip()
893 citation_content = data[1]
894 if key == "citation_author":
895 authors_strings.append(get_author_xml(template_str=citation_content))
896 continue
897 elif not authors_parsed:
898 xml_string += ", ".join(authors_strings)
899 authors_parsed = True
901 if key in cls.references_mapping:
902 xml_string += " " + cls.references_mapping[key](citation_content)
904 return parse_mixed_citation_into_ref(xml_string, label=label)
906 @classmethod
907 def get_or_create_source(cls):
908 source, created = Source.objects.get_or_create(
909 domain=cls.source_domain,
910 defaults={
911 "name": cls.source_name,
912 "website": cls.source_website,
913 "view_id": cls.get_view_id(),
914 },
915 )
916 if created:
917 source.save()
918 return source
920 @staticmethod
921 def get_issue_pid(
922 collection_id: str,
923 year: int | str,
924 volume_number: str | None = None,
925 issue_number: str | None = None,
926 series: str | None = None,
927 ):
928 # Replace any non-word character with an underscore
929 pid = f"{collection_id}_{year}"
930 if series is not None: 930 ↛ 931line 930 didn't jump to line 931 because the condition on line 930 was never true
931 pid += f"_{series}"
932 if volume_number is not None:
933 pid += f"_{volume_number}"
934 if issue_number is not None:
935 pid += f"_{issue_number}"
936 pid = regex.sub(r"[^\w-]+", "_", cleanup_str(pid))
937 return pid
939 @staticmethod
940 def set_pages(article: ArticleData, pages: str, separator: str = "-"):
941 pages_split = pages.split(separator)
942 if len(pages_split) == 0: 942 ↛ 943line 942 didn't jump to line 943 because the condition on line 942 was never true
943 article.page_range = pages
944 if len(pages_split) > 0: 944 ↛ exitline 944 didn't return from function 'set_pages' because the condition on line 944 was always true
945 if pages[0].isnumeric(): 945 ↛ exitline 945 didn't return from function 'set_pages' because the condition on line 945 was always true
946 article.fpage = pages_split[0]
947 if ( 947 ↛ 952line 947 didn't jump to line 952 because the condition on line 947 was never true
948 len(pages_split) > 1
949 and pages_split[0] != pages_split[1]
950 and pages_split[1].isnumeric()
951 ):
952 article.lpage = pages_split[1]
954 @staticmethod
955 def _process_pdf_header(chunk: str, response: requests.Response | aiohttp.ClientResponse):
956 content_type = response.headers.get("Content-Type")
957 if regex.match(rb"^%PDF-\d\.\d", chunk):
958 if content_type and "application/pdf" in content_type:
959 # The file is unmistakably a pdf
960 return [
961 True,
962 response,
963 {
964 "status": ExtlinkChecked.Status.OK,
965 "message": "",
966 },
967 ]
968 # The file is a pdf, but the content type advertised by the server is wrong
969 return [
970 True,
971 response,
972 {
973 "status": ExtlinkChecked.Status.WARNING,
974 "message": f"Content-Type header: {content_type}",
975 },
976 ]
978 # Reaching here means we couldn't find the pdf.
979 if not content_type or "application/pdf" not in content_type:
980 return [
981 False,
982 response,
983 {
984 "status": ExtlinkChecked.Status.ERROR,
985 "message": f"Content-Type header: {content_type}; PDF Header not found: got {chunk}",
986 },
987 ]
989 return [
990 False,
991 response,
992 {
993 "status": ExtlinkChecked.Status.ERROR,
994 "message": f"PDF Header not found: got {chunk}",
995 },
996 ]
998 @classmethod
999 async def a_check_pdf_link_validity(
1000 cls, url: str, verify=True
1001 ) -> list[bool | aiohttp.ClientResponse | dict]:
1002 """
1003 Check the validity of the PDF links.
1004 """
1005 CHUNK_SIZE = 10 # Nombre de caractères à récupérer
1006 headers = {
1007 "Range": f"bytes=0-{CHUNK_SIZE}",
1008 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0",
1009 }
1010 async with cls.async_session.get(
1011 url, headers=headers, allow_redirects=True, ssl=verify
1012 ) as response:
1013 try:
1014 chunk = await response.content.read(CHUNK_SIZE)
1015 return BaseCollectionCrawler._process_pdf_header(chunk, response)
1016 except StopIteration:
1017 return [
1018 False,
1019 response,
1020 {
1021 "status": ExtlinkChecked.Status.ERROR,
1022 "message": "Error reading PDF header",
1023 },
1024 ]
1026 @classmethod
1027 def check_pdf_link_validity(
1028 cls, url: str, verify=True
1029 ) -> list[bool | requests.Response | None | dict]:
1030 """
1031 Check the validity of the PDF links.
1032 """
1033 CHUNK_SIZE = 10 # Nombre de caractères à récupérer
1034 header = {
1035 "Range": f"bytes=0-{CHUNK_SIZE}",
1036 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0",
1037 }
1038 with get_session().get(
1039 url, headers=header, allow_redirects=True, verify=verify, stream=True
1040 ) as response:
1041 try:
1042 chunk = next(response.iter_content(CHUNK_SIZE))
1043 return BaseCollectionCrawler._process_pdf_header(chunk, response)
1044 except StopIteration:
1045 return [
1046 False,
1047 response,
1048 {
1049 "status": ExtlinkChecked.Status.ERROR,
1050 "message": "Error reading PDF header",
1051 },
1052 ]
1054 @classmethod
1055 async def check_extlink_validity(cls, extlink: "ExtLink"):
1056 """
1057 Method used by rot_monitoring to check if links have expired
1058 """
1059 defaults: dict = {"date": datetime.now(), "status": ExtlinkChecked.Status.OK}
1060 header = {
1061 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:140.0) Gecko/20100101 Firefox/140.0"
1062 }
1063 verify = True
1064 if not cls.verify:
1065 verify = False
1066 try:
1067 # For the GDZ links, we just check if the http response is 200 or 206
1068 if (
1069 extlink.rel == "article-pdf"
1070 and "gdz.sub.uni-goettingen.de" not in extlink.location
1071 ):
1072 isok, response, message = await cls.a_check_pdf_link_validity(
1073 extlink.location, verify
1074 )
1075 defaults.update(message)
1076 defaults["http_status"] = response.status
1077 else:
1078 async with cls.async_session.get(
1079 url=extlink.location,
1080 headers=header,
1081 allow_redirects=True,
1082 ssl=verify,
1083 ) as response:
1084 defaults["http_status"] = response.status
1085 if response.status not in (200, 206):
1086 defaults["status"] = ExtlinkChecked.Status.ERROR
1088 except aiohttp.ClientSSLError:
1089 cls.logger.error("SSL error for the url: %s", extlink.location)
1090 defaults["status"] = ExtlinkChecked.Status.ERROR
1091 defaults["message"] = "SSL error"
1092 except aiohttp.ClientConnectionError:
1093 cls.logger.error("Connection error for the url: %s", extlink.location)
1094 defaults["status"] = ExtlinkChecked.Status.ERROR
1095 defaults["message"] = "Connection error"
1096 except TimeoutError:
1097 cls.logger.error("Timeout error for the url: %s", extlink.location)
1098 defaults["status"] = ExtlinkChecked.Status.ERROR
1099 defaults["message"] = "Timeout error"
1100 finally:
1101 try:
1102 await ExtlinkChecked.objects.aupdate_or_create(extlink=extlink, defaults=defaults)
1103 cls.logger.info(
1104 "DB Update, source: %s, url: %s", cls.source_domain, extlink.location
1105 )
1106 except IntegrityError:
1107 cls.logger.error(
1108 "Extlink was deleted, source: %s, url: %s", cls.source_domain, extlink.location
1109 )
1111 def resolve_year_end(self, pid: str, default: int) -> int:
1112 if pid in self.pid_year_restrictions:
1113 return datetime.now().year - self.pid_year_restrictions[pid]
1114 return default
1117def get_first_last_years(years_str: str):
1118 years_splitted = years_str.split("-")
1119 fyear = int(years_splitted[0])
1120 lyear = None
1121 if len(years_splitted) > 1:
1122 lyear = int(years_splitted[1])
1123 return fyear, lyear