Coverage for src/crawler/by_source/dml_e_crawler.py: 28%
173 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1from urllib.parse import urljoin
3import regex
4from bs4 import BeautifulSoup, Tag
5from ptf.model_data import (
6 ArticleData,
7 IssueData,
8 create_articledata,
9 create_contributor,
10 create_extid,
11 create_extlink,
12)
14from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
15from crawler.crawler_utils import article_has_source
16from crawler.models import ExtlinkChecked
17from crawler.utils import add_pdf_link_to_xarticle
20class Dml_eCrawler(MatchingCrawler):
21 """
22 DML_E is quite peculiar :
23 There is no issue page, and articles are separated into "years" instead of volumes/issues.
24 volume/issue number is stored inside each article page.
25 In order to being able to parse volume and issue numbers, we must parse the articles before creating volumes and issues.
26 """
28 source_domain = "DML_E"
29 source_name = "Proyecto DML-E: Biblioteca Digital de Matemáticas "
30 source_website = "http://dmle.icmat.es/"
32 # 1987, 1: 1-17
33 # 1999,19: 1-11
34 # 2008, 53-62,
35 # 1963 (1-2):
36 # 2000, 51 (1): 49-58, 13 Ref.
37 # 2006, 57 (Extra): 327-342, 10 Ref.
38 issue_regex = r"\d+,? ?(?:(?P<volume>\d+),? ?)?(?:\((?P<number>[\d\w\-]+)\))?(?:[:,])? ?(?:(?P<page_start>\d+)-(?P<page_end>\d+))?"
40 def parse_collection_content(self, content):
41 xissues = []
42 soup = BeautifulSoup(content, "html.parser")
43 pagination_elements = soup.select("div.prevnext a")
44 for page in pagination_elements:
45 href = page.get("href")
46 if not isinstance(href, str): 46 ↛ 47line 46 didn't jump to line 47 because the condition on line 46 was never true
47 continue
48 href = urljoin(self.collection_url, href)
49 content = self.download_file(href)
50 xissues = [*xissues, *self.parse_collection_page(content, href)]
52 return xissues
54 def parse_collection_page(self, content: str, url: str):
55 soup = BeautifulSoup(content, "html.parser")
56 xissues = []
57 current_year = False
58 issues_tags = soup.select("a[name], ul.art_info")
59 for issue_tag in issues_tags:
60 if issue_tag.name == "a":
61 current_year = issue_tag.get("name")
62 if not isinstance(current_year, str): 62 ↛ 63line 62 didn't jump to line 63 because the condition on line 62 was never true
63 raise ValueError("Issue year cannot be parsed")
64 continue
66 if not current_year: 66 ↛ 67line 66 didn't jump to line 67 because the condition on line 66 was never true
67 raise ValueError("Issue year not found")
68 issue = self.create_xissue(url, int(current_year), current_year)
69 self.parse_issue_tag(issue_tag, issue)
70 xissues.append(issue)
71 return xissues
73 # def parse_issue_content(self, content, xissue):
74 # pass
76 def parse_issue_tag(self, tag: Tag, xissue: IssueData):
77 if not xissue.url: 77 ↛ 78line 77 didn't jump to line 78 because the condition on line 77 was never true
78 raise ValueError("xissue must have an URL")
79 article_tags = tag.select("li")
80 for index, art_tag in enumerate(article_tags):
81 href_tag = art_tag.select_one("a[href]")
82 if not href_tag: 82 ↛ 83line 82 didn't jump to line 83 because the condition on line 82 was never true
83 raise ValueError("Cannot parse article")
84 url = href_tag.get("href")
85 if not isinstance(url, str): 85 ↛ 86line 85 didn't jump to line 86 because the condition on line 85 was never true
86 raise ValueError("Cannot parse Article URL")
87 url = urljoin(xissue.url, url)
89 title = href_tag.text
91 article = create_articledata()
92 article.title_tex = title
93 article.url = url
94 article.pid = "a" + str(index)
95 xissue.articles.append(article)
97 def parse_dml_e_article_content(self, content, xissue, xarticle: ArticleData, url, pid):
98 xarticle.pid = pid
99 soup = BeautifulSoup(content, "html.parser")
100 table_lines = soup.select("div#centro table tr")
101 issue_volume: str | None = None
102 issue_number: str | None = None
103 for line in table_lines:
104 header_tag = line.select_one("th")
105 value_tag = line.select_one("td")
106 if not value_tag:
107 raise ValueError("Cannot parse article")
109 # PDF
110 if not header_tag:
111 href_tag = line.select_one("a")
112 if not href_tag:
113 raise ValueError("Cannot parse article pdf link")
114 href = href_tag.get("href")
115 if not isinstance(href, str):
116 raise ValueError("Cannot parse article pdf link")
117 add_pdf_link_to_xarticle(xarticle, self.source_website + href)
118 continue
120 # Title
121 if header_tag.text == "Título español":
122 xarticle.title_tex = value_tag.text
123 continue
124 if header_tag.text == "Título original":
125 xarticle.title_tex = value_tag.text
126 continue
127 if header_tag.text == "Título inglés":
128 xarticle.title_tex = value_tag.text
129 continue
131 # Author
132 if header_tag.text == "Autor/es":
133 authors_tags = value_tag.select("a")
134 for a in authors_tags:
135 author = create_contributor()
136 author["role"] = "author"
137 author["string_name"] = a.text
138 xarticle.contributors.append(author)
139 continue
140 # Page
141 if header_tag.text == "Publicación":
142 volume_re = list(regex.finditer(self.issue_regex, value_tag.text))
143 if len(volume_re) != 0:
144 # raise ValueError("Cannot parse Article page")
145 volume_data = volume_re[0].groupdict()
147 if volume_data["page_start"] and volume_data["page_end"]:
148 xarticle.page_range = (
149 volume_data["page_start"] + "-" + volume_data["page_end"]
150 )
151 if "volume" in volume_data:
152 issue_volume = volume_data["volume"]
153 if "number" in volume_data:
154 issue_number = volume_data["number"]
155 else:
156 raise ValueError("issue volume or number not found")
158 # LANG
159 if header_tag.text == "Idioma":
160 languages = {"Inglés": "en", "Español": "es", "Francés": "fr"}
161 if value_tag.text in languages:
162 xarticle.lang = languages[value_tag.text]
164 if header_tag.text == "Código MathReviews":
165 if value_tag.text.startswith("MR"):
166 extid = create_extid("mr-item-id", value_tag.text)
167 xarticle.extids.append(extid)
168 if header_tag.text == "Código Z-Math":
169 if value_tag.text.startswith("Zbl "):
170 # Space in zblid... http://dmle.icmat.es/revistas/detalle.php?numero=95
171 zblid = value_tag.text.removeprefix("Zbl ").strip()
172 # zblid does not seems to exist http://dmle.icmat.es/revistas/detalle.php?numero=4004
173 if not zblid.startswith("pre"):
174 extid = create_extid("zbl-item-id", zblid)
175 xarticle.extids.append(extid)
177 return xarticle, issue_volume, issue_number
179 def crawl_issue(self, xissue: IssueData):
180 if hasattr(xissue, "url") and xissue.url:
181 content = self.download_file(xissue.url)
182 self.parse_issue_content(content, xissue)
184 dml_e_issues: dict[str, IssueData] = {}
186 xarticles = xissue.articles
188 for xarticle in xarticles:
189 parsed_xarticle, xissue_vol, xissue_number = self.crawl_dml_e_article(xarticle, xissue)
190 if parsed_xarticle is None:
191 continue
192 if xissue_vol or xissue_number:
193 issue_tag = (xissue_vol or "") + "_" + (xissue_number or "")
194 else:
195 issue_tag = xissue.fyear
196 if not issue_tag:
197 raise ValueError("issue_tag is None")
198 if issue_tag not in dml_e_issues:
199 dml_e_issues[issue_tag] = self.create_xissue(
200 xissue.url, xissue.fyear, xissue_vol, xissue_number or None
201 )
202 dml_e_issues[issue_tag].articles.append(parsed_xarticle)
204 for value in dml_e_issues.values():
205 if self.ignore_missing_pdf:
206 value.articles = [a for a in value.articles if self.article_has_pdf(a)]
207 if self.dry:
208 return
209 issue_has_pdf = self.article_has_pdf(value)
210 if len(value.articles) == 0 and not issue_has_pdf:
211 continue
212 for index, article in enumerate(value.articles):
213 article.pid = f"{value.pid}_a{index}"
214 self.process_resource_metadata(value, resource_type="issue")
215 self.add_xissue_into_database(value)
217 def crawl_dml_e_article(self, xarticle: ArticleData, xissue: IssueData):
218 parsed_xarticle = xarticle
219 if not hasattr(xarticle, "url") or not xarticle.url:
220 raise ValueError("article does not have an url")
221 # self.progress_bar.text(f"{xarticle.pid} - {xarticle.url}")
223 content = self.download_file(xarticle.url)
224 pid = f"{xissue.pid}_{xarticle.pid}"
226 parsed_xarticle, xissue_vol, xissue_number = self.parse_dml_e_article_content(
227 content, xissue, xarticle, xarticle.url, pid
228 )
230 if not article_has_source(parsed_xarticle) and parsed_xarticle.url:
231 ext_link = create_extlink()
232 ext_link["rel"] = "source"
233 ext_link["location"] = parsed_xarticle.url
234 ext_link["metadata"] = self.source_domain
235 parsed_xarticle.ext_links.append(ext_link)
237 # The article title may have formulas surrounded with '$'
238 return self.process_article_metadata(parsed_xarticle), xissue_vol, xissue_number
240 @classmethod
241 def check_pdf_link_validity(cls, url, verify=True):
242 # we overwrite this base_crawler method to manage the links to pdf that are not article pdf.
243 # Avoid downloading the whole PDF
244 # CHUNK_SIZE = 100 # number of characters fetched
245 # If the url contains Movingwall it does not lead to the article
246 # TODO this should be in the harvest tasks
247 if "Movingwall" in url:
248 print("The url does not link to the PDF article because of o moving wall")
249 return (
250 False,
251 None,
252 {
253 "status": ExtlinkChecked.Status.ERROR,
254 "message": "The url does not link to the PDF article because of a moving wall",
255 },
256 )
257 return super().check_pdf_link_validity(url)