Coverage for src/crawler/utils.py: 40%
144 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import base64
2import hashlib
3import json
4import logging
5import os
6import pickle
7import re
8import unicodedata
9from collections.abc import Callable
10from datetime import datetime, timezone
11from functools import lru_cache, wraps
12from typing import Literal, TypeVar
14import regex
15from bs4 import BeautifulSoup
16from django.conf import settings
17from django.contrib.auth.models import User
18from history.model_data import HistoryEventDict, HistoryEventType
19from history.models import HistoryEventStatus
21# from ptf.models import ResourceId
22from history.utils import insert_history_event
23from ptf import model_helpers
24from ptf.cmds import ptf_cmds
25from ptf.cmds.xml.xml_utils import escape
26from ptf.exceptions import ResourceDoesNotExist
27from ptf.model_data import ResourceData, create_extlink, create_publicationdata
28from ptf.models import Collection
29from pymongo import MongoClient
30from pymongo.synchronous.database import Database
32from crawler.types import JSONCol
34logger = logging.getLogger(__name__)
37def insert_crawl_event_in_history(
38 colid: str,
39 source_domain: str,
40 username: str,
41 status: HistoryEventStatus,
42 tasks_count,
43 message: str,
44 event_type: HistoryEventType = "import-collection",
45 title=None,
46):
47 collection = model_helpers.get_collection(colid, sites=False)
48 user = User.objects.get(username=username)
50 event_data: HistoryEventDict = {
51 "type": event_type,
52 "pid": f"{colid}-{source_domain}",
53 "col": collection,
54 "source": source_domain,
55 "status": status,
56 "title": collection.title_html if collection else (title or colid),
57 "userid": user.pk,
58 "type_error": "",
59 "message": message,
60 }
62 insert_history_event(event_data)
65def col_has_source(col: JSONCol, filter: str):
66 return any(source for source in col["sources"] if source == filter)
69def get_cols_by_source(source: str) -> list[JSONCol]:
70 """
71 Get all cols by source
72 @param source: str
73 @return: list of collections
74 """
75 data = get_all_cols()
77 return [col for col in data.values() if col_has_source(col, source)]
80def get_all_cols_by_source():
81 """
82 Get all cols by source
83 @return: dict of collections by source
84 """
85 data = get_all_cols()
87 sources: dict[str, list] = {}
88 for col in data.values():
89 for source in col["sources"]:
90 if source not in sources:
91 sources[source] = []
92 sources[source].append(col)
94 return sources
97@lru_cache(maxsize=None)
98def get_all_cols() -> dict[str, JSONCol]:
99 with open(
100 os.path.dirname(os.path.abspath(__file__)) + "/data/all_cols.json", encoding="utf8"
101 ) as data_collections:
102 return json.load(data_collections)
105def get_or_create_collection(pid: str):
106 """
107 Creates a Collection based on its pid.
108 The pid has to be in the list of collections given by the Documentation team (CSV then transformed in JSON)
109 """
111 all_collections = get_all_cols()
113 if pid not in all_collections:
114 raise ValueError(f"{pid} is not listed in all_cols.csv")
116 col_data = [item for item in all_collections.items() if item[0] == pid][0][1]
118 collection: Collection | None = model_helpers.get_collection(pid, sites=False)
120 if not collection:
121 p = model_helpers.get_provider("mathdoc-id")
123 xcol = create_publicationdata()
124 xcol.coltype = col_data["type"]
125 xcol.pid = pid
126 xcol.title_tex = col_data["title"]
127 # Mis en commentaire car trop tôt, Taban n'a pas encore validé les ISSNs
128 # xcol.e_issn = col_data["ISSN_électronique"]
129 # xcol.issn = col_data["ISSN_papier"]
130 xcol.title_html = col_data["title"]
131 xcol.title_xml = f"<title-group><title>{col_data['title']}</title></title-group>"
132 xcol.lang = "en"
134 cmd = ptf_cmds.addCollectionPtfCmd({"xobj": xcol})
135 cmd.set_provider(p)
136 collection = cmd.do()
138 # Mis en commentaire car trop tôt, Taban n'a pas encore validé les ISSNs
139 # if col_data["ISSN_électronique"] != "":
140 # e_issn = {
141 # "resource_id": collection.resource_ptr_id,
142 # "id_type": "e_issn",
143 # "id_value": col_data["ISSN_électronique"],
144 # }
145 # ResourceId.objects.create(**e_issn)
146 #
147 # if col_data["ISSN_papier"] != "":
148 # issn = {
149 # "resource_id": collection.resource_ptr_id,
150 # "id_type": "issn",
151 # "id_value": col_data["ISSN_papier"],
152 # }
153 # ResourceId.objects.create(**issn)
155 if not collection:
156 raise ResourceDoesNotExist(f"Resource {pid} does not exist")
158 return collection
161def cleanup_str(input: str, unsafe=False):
162 # some white spaces aren't actual space characters, like \xa0
163 input = unicodedata.normalize("NFKC", input)
164 #
165 input = re.sub(r"[\x7f]+", "", input)
166 # remove useless continuous \n and spaces from the string
167 input = re.sub(r"[\n\t\r ]+", " ", input).strip()
168 if unsafe: 168 ↛ 169line 168 didn't jump to line 169 because the condition on line 168 was never true
169 return input
170 return escape(input)
173def add_pdf_link_to_xarticle(
174 xarticle: ResourceData,
175 pdf_url: str,
176 mimetype: Literal["application/pdf", "application/x-tex", "text/html"] = "application/pdf",
177):
178 xarticle.streams.append(
179 {
180 "rel": "full-text",
181 "mimetype": mimetype,
182 "location": pdf_url,
183 "base": "",
184 "text": "Full Text",
185 }
186 )
188 # The pdf url is already added as a stream (just above) but might be replaced by a file later on.
189 # Keep the pdf url as an Extlink if we want to propose both option:
190 # - direct download of a local PDF
191 # - URL to the remote PDF
192 if mimetype == "application/pdf": 192 ↛ 194line 192 didn't jump to line 194 because the condition on line 192 was always true
193 rel = "article-pdf"
194 elif mimetype == "text/html":
195 rel = "article-html"
196 else:
197 rel = "article-tex"
198 ext_link = create_extlink(rel=rel, location=pdf_url)
199 xarticle.ext_links.append(ext_link)
202def add_source_link_to_xarticle(xarticle, url: str, domain: str):
203 ext_link = create_extlink()
204 ext_link["rel"] = "source"
205 ext_link["location"] = url
206 ext_link["metadata"] = domain
207 xarticle.ext_links.append(ext_link)
210def regex_to_dict(pattern: str, value: str, *, error_msg="Regex failed to parse"):
211 issue_search = regex.search(pattern, value)
212 if not issue_search: 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true
213 raise ValueError(error_msg)
215 return issue_search.groupdict()
218def get_base(soup: BeautifulSoup, default: str):
219 base_tag = soup.select_one("head base")
220 if not base_tag:
221 return default
222 base = base_tag.get("href")
223 if not isinstance(base, str):
224 raise ValueError("Cannot parse base href")
225 return base
228try:
229 from crawler.tests.data_generation.decorators import skip_generation
230except ImportError:
232 def skip_generation(func):
233 def wrapper(*args, **kwargs):
234 return func(*args, **kwargs)
236 return wrapper
239class PickleSerializer:
240 def serialize(self, obj):
241 return base64.b64encode(pickle.dumps(obj, protocol=-1))
243 def deserialize(self, serialized):
244 return pickle.loads(base64.b64decode(serialized))
247RT = TypeVar("RT")
250def mongo_cache(
251 db_conn: Database = MongoClient(host=getattr(settings, "MONGO_HOSTNAME", "localhost"))[
252 "crawler_func_cache"
253 ],
254 prefix="cache_",
255 capped=True,
256 capped_size=1000000000,
257 hash_keys=True,
258 serializer=PickleSerializer,
259) -> Callable[[Callable[..., RT]], Callable[..., RT]]:
260 """Helper decorator to speedup local development
261 ```py
264 @mongo_cache(
265 db_conn=MongoClient(host=getattr(settings, "MONGO_HOSTNAME", "localhost"))[
266 "crawler_func_cache"
267 ]
268 )
269 def parse_collection_content(self, content):
270 ...
271 ```
272 """
274 def decorator(func: Callable[..., RT]) -> Callable[..., RT]:
275 serializer_ins = serializer()
276 col_name = f"{prefix}{func.__name__}"
277 if capped:
278 db_conn.create_collection(col_name, capped=capped, size=capped_size)
280 cache_col = db_conn[col_name]
281 cache_col.create_index("key", unique=True)
282 cache_col.create_index("date", expireAfterSeconds=86400)
284 @wraps(func)
285 def wrapped_func(*args, **kwargs) -> RT:
286 cache_key = pickle.dumps((args[1:], kwargs), protocol=-1)
287 if hash_keys:
288 cache_key = hashlib.md5(cache_key).hexdigest()
289 else:
290 cache_key = base64.b64encode(cache_key)
292 cached_obj = cache_col.find_one(dict(key=cache_key))
293 if cached_obj:
294 return serializer_ins.deserialize(cached_obj["result"])
296 ret = func(*args, **kwargs)
297 cache_col.update_one(
298 {"key": cache_key},
299 {
300 "$set": {
301 "result": serializer_ins.serialize(ret),
302 "date": datetime.now(timezone.utc),
303 }
304 },
305 upsert=True,
306 )
308 return ret
310 return wrapped_func
312 return decorator