Coverage for src/crawler/by_source/episciences_crawlers/episciences_epidemes_crawler.py: 20%
47 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import json
2import logging
4from crawler.by_source.episciences_crawlers.episciences_crawler import EpisciencesCrawler
5from crawler.utils import regex_to_dict
7# TODO : Find another system to save the title information with
8# Maybe a toggle attribute on collection ?
10HAS_TITLES = {"COMPOSIT": False, "ARIMA": True}
12logger = logging.getLogger(__name__)
15# We could improve our data further by augmenting the articles using arxiv
16# (references)
17class EpisciencesEpidemesCrawler(EpisciencesCrawler):
18 source_domain = "EPISCIENCES_EPIDEMES"
20 volume_title_re = r"(?P<number>\d+)\s*\|\s*(?P<year>\d+)"
21 article_volume_match_re = r"(?P<number>\d+)\s*\|\s*(?P<year>\d+)"
23 def process_issue_title(self, xissue: dict, issue_title: str):
24 title_dict = regex_to_dict(
25 self.volume_title_re,
26 issue_title,
27 error_msg="Couldn't parse issue title",
28 )
30 xissue.volume = title_dict["number"]
31 xissue.title_tex = f"Volume {xissue.volume}"
32 if title_dict["year"]:
33 xissue.year = title_dict["year"]
34 return xissue
36 def parse_collection_content(self, content):
37 data = json.loads(content)
38 xissues = []
39 volume = len(data)
40 for issue in data:
41 # if "special_issue" in issue.get("vol_type"):
42 # continue
43 url = f"https://api.episciences.org/api/volumes/{issue['vid']}?rvcode={self.episciences_id}&pagination=false"
44 issue_content = json.loads(self.download_file(url))
45 # Would a dedicated class be preferable ? Or maybe a smarter crawler overall
46 xissue = self.prefetch_episciences_issue(issue_content)
47 xissue.volume = str(volume)
48 xissues.append(xissue)
49 volume -= 1
51 xissues.sort(key=lambda x: x.volume)
52 issues_by_volume = {}
53 for issue in xissues:
54 if issue.volume not in issues_by_volume:
55 issues_by_volume[issue.volume] = []
56 issues_by_volume[issue.volume].append(issue)
58 for volume_issues in issues_by_volume.values():
59 year_iterable = []
60 try:
61 year_iterable = [int(i.fyear) for i in volume_issues]
62 except ValueError:
63 pass
64 firstyear = min(year_iterable)
65 lastyear = max(year_iterable)
66 if firstyear != lastyear:
67 for i in volume_issues:
68 i.fyear = firstyear
69 i.lyear = lastyear
71 return xissues