lms.util.net

Utilities for network and HTTP.

  1"""
  2Utilities for network and HTTP.
  3"""
  4
  5import re
  6import typing
  7import urllib.parse
  8
  9import bs4
 10import edq.net.exchange
 11import edq.util.hash
 12import edq.util.json
 13import requests
 14
 15import lms.model.constants
 16
 17STANDARDIZED_TIMESTAMP: str = '123456789'
 18STANDARDIZED_SESSION_KEY: str = 'abcABC123'
 19STANDARDIZED_RANDOM_STRING: str = 'abc123'
 20
 21CANVAS_CLEAN_REMOVE_CONTENT_KEYS: typing.List[str] = [
 22    'created_at',
 23    'ics',
 24    'last_activity_at',
 25    'lti_context_id',
 26    'preview_url',
 27    'secure_params',
 28    'total_activity_time',
 29    'updated_at',
 30    'url',
 31    'uuid',
 32]
 33""" Keys to remove from Canvas content. """
 34
 35BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS: typing.List[str] = [
 36    'created',
 37    'modified',
 38]
 39""" Keys to remove from Blackboard content. """
 40
 41BLACKBOARD_CLEAN_REMOVE_HEADERS: typing.Set[str] = {
 42    'access-control-allow-origin',
 43    'content-encoding',
 44    'content-language',
 45    'expires',
 46    'last-modified',
 47    'p3p',
 48    'strict-transport-security',
 49    'transfer-encoding',
 50    'vary',
 51    'x-blackboard-xsrf',
 52}
 53""" Keys to remove from Blackboard headers. """
 54
 55MOODLE_CLEAN_REMOVE_HEADERS: typing.Set[str] = {
 56    'accept-ranges',
 57    'content-encoding',
 58    'content-language',
 59    'content-script-type',
 60    'content-style-type',
 61    'expires',
 62    'keep-alive',
 63    'last-modified',
 64    'transfer-encoding',
 65    'vary',
 66}
 67""" Keys to remove from Moodle headers. """
 68
 69MOODLE_FINALIZE_REMOVE_PARAMS: typing.Set[str] = {
 70    'logintoken',
 71}
 72""" Keys to remove from Moodle headers. """
 73
 74MOODLE_HTML_CLEAN: typing.Dict[str, typing.Dict[str, typing.Any]] = {
 75    r'/login/index\.php': {
 76        'delete_elements': ['script', 'footer'],
 77        'remove_all_attrs': ['body', 'div'],
 78        'hoist_elements': {
 79            ('div#page-wrapper', 'input[name="logintoken"]'),
 80        },
 81    },
 82    r'/user/index\.php\?id=(\d+)': {
 83        'delete_elements': [
 84            'tr.emptyrow',
 85            'div[data-status="Active"]',
 86            'i',
 87            'th.a',
 88        ],
 89        'attrs_to_keep': {
 90            'a': ['data-column'],
 91            'th': ['class'],
 92            'td': ['class'],
 93        },
 94        'remove_all_attrs': ['tr'],
 95        'final_selectors': ['table#participants'],
 96    },
 97    r'/user/profile.php': {
 98        'filter_elements_by_descendant': {
 99            'div.card-body': ('h3', 'Course details'),
100        },
101        'hoist_elements': {
102            ('div.card-body ul', 'div.card-body ul li ul'),
103        },
104        'final_selectors': ['div.card-body'],
105    },
106    r'/grade/report/grader/index\.php\?id=(\d+)': {
107        'delete_elements': [
108            'caption',
109            'tr[class=""]',
110            'tr[class="avg"]',
111            'i',
112            'button',
113            'a.dropdown-item',
114            'img',
115            'label',
116        ],
117        'attrs_to_keep': {
118            'input': ['max', 'name', 'data-context'],
119            'a': ['class'],
120            'th': ['class', 'data-itemid'],
121            'td': ['class'],
122            'td input': ['max'],
123        },
124        'classes_to_keep': {
125            'th': ['item', re.compile(r'c\d+')],
126            'td': [re.compile(r'c\d+')],
127        },
128        'remove_all_attrs': [
129            'div',
130            'tr',
131        ],
132        'hoist_elements': {
133            ('div.d-flex.flex-column.h-100', 'a.gradeitemheader'),
134            ('div.d-flex.flex-column.h-100', 'input[title="Grade"]'),
135        },
136        'replacements': [
137            (r'\n', ''),
138            (r'\s+aria-(?:label|hidden|expanded)="[^"]*"', ''),
139            (r'\s+role\s*=\s*(?:"[^"]*"|\'[^\']*\')', ''),
140            (r'<div>', ''),
141            (r'</div>', ''),
142            (r'(<script>)[\s\S]*?(</script>)', rf'\1M.cfg = {{"sesskey":"{STANDARDIZED_SESSION_KEY}"}}\2')
143        ],
144        'final_selectors': ['head script:not([type])', 'table#user-grades', 'input[name=setmode]'],
145    },
146}
147""" A mapping of Moodle URL patterns to clean_html() kwargs. """
148
149def clean_lms_response(response: requests.Response, body: str) -> str:
150    """
151    A ResponseModifierFunction that attempt to identify
152    if the requests comes from a Learning Management System (LMS),
153    and clean the response accordingly.
154    """
155
156    # Check the standard LMS Toolkit backend header.
157    raw_backend_type = response.headers.get(lms.model.constants.HEADER_KEY_BACKEND, '').lower()
158
159    if (raw_backend_type == lms.model.constants.BackendType.CANVAS.value):
160        return clean_canvas_response(response, body)
161
162    if (raw_backend_type == lms.model.constants.BackendType.MOODLE.value):
163        return clean_moodle_response(response, body)
164
165    # Try looking inside the header keys.
166    for key in response.headers:
167        key = key.lower().strip()
168
169        if ('blackboard' in key):
170            return clean_blackboard_response(response, body)
171
172        if ('canvas' in key):
173            return clean_canvas_response(response, body)
174
175        if ('moodle' in key):
176            return clean_moodle_response(response, body)
177
178    return body
179
180def clean_blackboard_response(response: requests.Response, body: str) -> str:
181    """
182    See clean_lms_response(), but specifically for the Blackboard LMS.
183    This function will:
184     - Call _clean_base_response().
185     - Remove specific headers.
186    """
187
188    body = _clean_base_response(response, body)
189
190    # Work on both request and response headers.
191    remove_headers(response, BLACKBOARD_CLEAN_REMOVE_HEADERS)
192
193    # Most blackboard responses are JSON.
194    try:
195        data = edq.util.json.loads(body, strict = True)
196    except Exception:
197        # Response is not JSON.
198        return body
199
200    # Remove any content keys.
201    _recursive_remove_keys(data, set(BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS))
202
203    # Convert body back to a string.
204    body = edq.util.json.dumps(data)
205
206    return body
207
208def clean_canvas_response(response: requests.Response, body: str) -> str:
209    """
210    See clean_lms_response(), but specifically for the Canvas LMS.
211    This function will:
212     - Call _clean_base_response().
213     - Remove content keys: [last_activity_at, total_activity_time]
214    """
215
216    body = _clean_base_response(response, body)
217    url = str(response.request.url)
218
219    if ('/files_api' in url):
220        # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing.
221        parts = urllib.parse.urlsplit(response.headers['location'])
222        parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '')
223        response.headers['location'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG)
224
225    # Most canvas responses are JSON.
226    try:
227        data = edq.util.json.loads(body, strict = True)
228    except Exception:
229        # Response is not JSON.
230        return body
231
232    # Remove any content keys.
233    _recursive_remove_keys(data, set(CANVAS_CLEAN_REMOVE_CONTENT_KEYS))
234
235    # Handle endpoint-specific cases.
236    if ('submissions/update_grades' in url):
237        data.pop('id', None)
238    elif (re.search(r'api/v1/courses/\w+/files', url) is not None):
239        # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing.
240        parts = urllib.parse.urlsplit(data['upload_url'])
241        parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '')
242        data['upload_url'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG)
243
244    # Convert body back to a string.
245    body = edq.util.json.dumps(data)
246
247    return body
248
249def finalize_canvas_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
250    """ Finalize Canvas exchanges. """
251
252    if (re.search(r'^api/v1/courses/\w+/files$', exchange.url_path) is not None):
253        # Clean out random data for the file upload.
254        data = edq.util.json.loads(str(exchange.response_body))
255
256        body_data = {
257            'upload_params': {
258                'Filename': data['upload_params']['Filename'],
259            },
260            'upload_url': data['upload_url'],
261        }
262
263        exchange.response_body = edq.util.json.dumps(body_data)
264    elif (exchange.url_path == 'files_api'):
265        # File upload calls contain several random pieces of information generated from the server.
266        # Drop information we don't need in testing and make the rest consistent.
267
268        filename = exchange.parameters['Filename']
269        exchange.parameters = {'filename': filename}
270        exchange.files = []
271
272        location = exchange.response_headers.get('location', None)
273        if (location is not None):
274            location = re.sub(r'uuid=.*$', f"uuid={edq.util.hash.sha256_hex(filename)}", str(location))
275
276        exchange.response_headers['location'] = location
277    elif (re.search(r'^api/v1/files/\w+/create_success$', exchange.url_path) is not None):
278        # Change the uuid parameter to match the one set in files_api (which redirects to this URL).
279        data = edq.util.json.loads(str(exchange.response_body))
280        exchange.parameters['uuid'] = edq.util.hash.sha256_hex(data['display_name'])
281        exchange.response_body = edq.util.json.dumps({'id': data['id']})
282
283    return exchange
284
285def clean_moodle_response(response: requests.Response, body: str) -> str:
286    """
287    See clean_lms_response(), but specifically for the Moodle LMS.
288    This function will:
289     - Call _clean_base_response().
290    """
291
292    body = _clean_base_response(response, body)
293
294    # Standardize timestamp.
295    current_timestamp_match = re.search(r"boost/theme/(\d{10})/favicon", body)
296    if (current_timestamp_match is not None):
297        body = body.replace(current_timestamp_match.group(1), STANDARDIZED_TIMESTAMP)
298
299    # Standardize session key.
300    session_key_match = re.search(r'"sesskey":"([^"]+)"', body)
301    if (session_key_match is not None):
302        body = body.replace(session_key_match.group(1), STANDARDIZED_SESSION_KEY)
303
304    # Standardize "random" string.
305    random_string_match = re.search(r"'random([a-z0-9]+)'", body)
306    if (random_string_match is not None):
307        body = body.replace(random_string_match.group(1), STANDARDIZED_RANDOM_STRING)
308
309    # Standardize logintoken.
310    logintoken_match = re.search(r'name="logintoken" value="(\w+)"', body)
311    if (logintoken_match is not None):
312        body = body.replace(logintoken_match.group(1), STANDARDIZED_SESSION_KEY)
313
314    # Standardize last access to course.
315    last_access_match = re.search(r'(\d+) secs', body)
316    if (last_access_match is not None):
317        body = body.replace(last_access_match.group(0), f'{STANDARDIZED_TIMESTAMP} secs')
318
319    # Work on both request and response headers.
320    remove_headers(response, MOODLE_CLEAN_REMOVE_HEADERS)
321
322    # Clean HTML responses.
323    for (pattern, clean_params) in MOODLE_HTML_CLEAN.items():
324        if (re.search(pattern, response.url.strip())):
325            body = clean_html(body, **clean_params)
326
327    return body
328
329def clean_html(
330        html: str,
331        filter_elements_by_descendant: typing.Union[typing.Dict[str, typing.Tuple[str, str]], None] = None,
332        delete_elements: typing.Union[typing.List[str], None] = None,
333        attrs_to_keep: typing.Union[typing.Dict[str, typing.List[str]], None] = None,
334        classes_to_keep: typing.Union[typing.Dict[str, typing.List[typing.Union[str, re.Pattern]]], None] = None,
335        remove_all_attrs: typing.Union[typing.List[str], None] = None,
336        hoist_elements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None,
337        replacements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None,
338        final_selectors: typing.Union[typing.List[str], None] = None,
339        ) -> str:
340    """
341    General purpose HTML cleaning function.
342
343    filter_elements_by_descendant: { selector: (descendant_selector, text), ... }
344    Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text.
345    For example:
346    ```
347    >>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>'
348    >>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')})
349    <div><h2>Hello World!</h2></div>
350    ```
351
352    delete_elements: [ selector, ... ]
353    Deletes any matching elements.
354
355    attrs_to_keep: { selector: [attribute, ...], ... }
356    Keeps only the listed attributes of elements matching the selector.
357
358    classes_to_keep: { selector: [class/pattern, ...], ... }
359    Keeps only listed classes, or classes matching the pattern, of elements matching the selector.
360
361    remove_all_attrs: [ selector, ... ]
362    Removes all attributes of matching elements.
363
364    hoist_elements: [ (parent_selector, child_selector), ... ]
365    For each pair of selectors, the parent element is replaced by the first matching child within the parent.
366    No replacement occurs if both matches are not found.
367
368    replacements: [ (pattern, replacement), ... ]
369    Performs a replacement for all regex matches.
370
371    final_selectors: [ selector, ... ]
372    Replaces html with all matched elements.
373    Useful for targeting only the necessary parts of the response.
374    Defaults to ['body'] if None.
375    """
376
377    if (filter_elements_by_descendant is None):
378        filter_elements_by_descendant = {}
379
380    if (delete_elements is None):
381        delete_elements = []
382
383    if (attrs_to_keep is None):
384        attrs_to_keep = {}
385
386    if (classes_to_keep is None):
387        classes_to_keep = {}
388
389    if (remove_all_attrs is None):
390        remove_all_attrs = []
391
392    if (hoist_elements is None):
393        hoist_elements = []
394
395    if (replacements is None):
396        replacements = [(r'\n', '')]
397
398    if (final_selectors is None):
399        final_selectors = ['body']
400
401    document = bs4.BeautifulSoup(html, 'html.parser')
402
403    # Filter Elements by Descendant
404    for (selector, (descendant_selector, text)) in filter_elements_by_descendant.items():
405        for element in document.select(selector):
406            descendant = element.select_one(descendant_selector)
407            if ((descendant is None) or (descendant.get_text() != text)):
408                element.decompose()
409
410    # Delete Elements
411    for selector in delete_elements:
412        for element in document.select(selector):
413            element.decompose()
414
415    # Keep Attributes
416    for (element_selector, attrs) in attrs_to_keep.items():
417        for element in document.select(element_selector):
418            # Remove extra attributes by keeping only select attributes and replacing the existing attribute dict.
419            element.attrs = {attr: element.attrs[attr] for attr in attrs if (attr in element.attrs)}
420
421    # Keep Classes
422    for (element_selector, keep_classes) in classes_to_keep.items():
423        for element in document.select(element_selector):
424            classes = element.get('class')
425            if ((classes is None) or (len(classes) == 0)):
426                continue
427
428            # Keep only classes listed for this selector.
429            kept = []
430            for keep_class in keep_classes:
431                for class_name in classes:
432                    if (((isinstance(keep_class, re.Pattern)) and (keep_class.fullmatch(class_name))) or (keep_class == class_name)):
433                        kept.append(class_name)
434
435            element['class'] = kept  # type: ignore[assignment]
436
437    # Remove All Attributes
438    for selector in remove_all_attrs:
439        for element in document.select(selector):
440            element.attrs.clear()
441
442    # Element Hoisting
443    for (parent_selector, child_selector) in hoist_elements:
444        for parent in document.select(parent_selector):
445            child = parent.select_one(child_selector)
446            if (child is None):
447                continue
448
449            parent.replace_with(child.extract())
450
451    # Final Selectors
452    elements = []
453    for final_selector in final_selectors:
454        for element in document.select(final_selector):
455            elements.append(element)
456
457    document_string = "".join([str(element) for element in elements])
458
459    # Replacements
460    for (pattern, replacement) in replacements:
461        document_string = re.sub(pattern, replacement, document_string)
462
463    return document_string
464
465def remove_headers(response: requests.Response, headers_to_remove: typing.Set[str]) -> None:
466    """
467    Remove headers from response and response's request.
468    """
469
470    for headers in [response.headers, response.request.headers]:
471        for key in list(headers.keys()):  # type: ignore[attr-defined]
472            if (key.strip().lower() in headers_to_remove):
473                headers.pop(key, None)  # type: ignore[attr-defined]
474
475def finalize_moodle_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
476    """ Finalize Moodle exchanges. """
477
478    for param in MOODLE_FINALIZE_REMOVE_PARAMS:
479        exchange.parameters.pop(param, None)
480
481    # Standardize session key.
482    if ('sesskey' in exchange.parameters):
483        exchange.parameters['sesskey'] = STANDARDIZED_SESSION_KEY
484
485    return exchange
486
487def _clean_base_response(response: requests.Response, body: str,
488        keep_headers: typing.Union[typing.List[str], None] = None) -> str:
489    """
490    Do response cleaning that is common amongst all backend types.
491    This function will:
492     - Remove X- headers.
493    """
494
495    # Index requests are generally for identification, and we use headers.
496    path = urllib.parse.urlparse(response.request.url).path.strip()
497    if (path in ['', '/']):
498        body = ''
499
500    for key in list(response.headers.keys()):
501        key = key.strip().lower()
502        if ((keep_headers is not None) and (key in keep_headers)):
503            continue
504
505        if (key.startswith('x-')):
506            response.headers.pop(key, None)
507
508    return body
509
510def _recursive_remove_keys(data: typing.Any, remove_keys: typing.Set[str]) -> None:
511    """
512    Recursively descend through the given and remove any instance to the given key from any dictionaries.
513    The data should only be simple types (POD, dicts, lists, tuples).
514    """
515
516    if (isinstance(data, (list, tuple))):
517        for item in data:
518            _recursive_remove_keys(item, remove_keys)
519    elif (isinstance(data, dict)):
520        for key in list(data.keys()):
521            if (key in remove_keys):
522                del data[key]
523            else:
524                _recursive_remove_keys(data[key], remove_keys)
525
526def parse_cookies(
527        text_cookies: typing.Union[str, None],
528        strip_key_prefix: bool = True,
529        ) -> typing.Dict[str, typing.Any]:
530    """ Parse cookies out of a text string. """
531
532    cookies: typing.Dict[str, typing.Any] = {}
533
534    if (text_cookies is None):
535        return cookies
536
537    text_cookies = text_cookies.strip()
538    if (len(text_cookies) == 0):
539        return cookies
540
541    for cookie in text_cookies.split('; '):
542        parts = cookie.split('=', maxsplit = 1)
543
544        key = parts[0].lower()
545
546        if (strip_key_prefix):
547            key = key.split(', ')[-1]
548
549        if (len(parts) == 1):
550            cookies[key] = True
551        else:
552            cookies[key] = parts[1]
553
554    return cookies
STANDARDIZED_TIMESTAMP: str = '123456789'
STANDARDIZED_SESSION_KEY: str = 'abcABC123'
STANDARDIZED_RANDOM_STRING: str = 'abc123'
CANVAS_CLEAN_REMOVE_CONTENT_KEYS: List[str] = ['created_at', 'ics', 'last_activity_at', 'lti_context_id', 'preview_url', 'secure_params', 'total_activity_time', 'updated_at', 'url', 'uuid']

Keys to remove from Canvas content.

BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS: List[str] = ['created', 'modified']

Keys to remove from Blackboard content.

BLACKBOARD_CLEAN_REMOVE_HEADERS: Set[str] = {'content-encoding', 'expires', 'content-language', 'strict-transport-security', 'access-control-allow-origin', 'transfer-encoding', 'vary', 'last-modified', 'p3p', 'x-blackboard-xsrf'}

Keys to remove from Blackboard headers.

MOODLE_CLEAN_REMOVE_HEADERS: Set[str] = {'keep-alive', 'content-encoding', 'expires', 'accept-ranges', 'content-language', 'content-style-type', 'transfer-encoding', 'vary', 'last-modified', 'content-script-type'}

Keys to remove from Moodle headers.

MOODLE_FINALIZE_REMOVE_PARAMS: Set[str] = {'logintoken'}

Keys to remove from Moodle headers.

MOODLE_HTML_CLEAN: Dict[str, Dict[str, Any]] = {'/login/index\\.php': {'delete_elements': ['script', 'footer'], 'remove_all_attrs': ['body', 'div'], 'hoist_elements': {('div#page-wrapper', 'input[name="logintoken"]')}}, '/user/index\\.php\\?id=(\\d+)': {'delete_elements': ['tr.emptyrow', 'div[data-status="Active"]', 'i', 'th.a'], 'attrs_to_keep': {'a': ['data-column'], 'th': ['class'], 'td': ['class']}, 'remove_all_attrs': ['tr'], 'final_selectors': ['table#participants']}, '/user/profile.php': {'filter_elements_by_descendant': {'div.card-body': ('h3', 'Course details')}, 'hoist_elements': {('div.card-body ul', 'div.card-body ul li ul')}, 'final_selectors': ['div.card-body']}, '/grade/report/grader/index\\.php\\?id=(\\d+)': {'delete_elements': ['caption', 'tr[class=""]', 'tr[class="avg"]', 'i', 'button', 'a.dropdown-item', 'img', 'label'], 'attrs_to_keep': {'input': ['max', 'name', 'data-context'], 'a': ['class'], 'th': ['class', 'data-itemid'], 'td': ['class'], 'td input': ['max']}, 'classes_to_keep': {'th': ['item', re.compile('c\\d+')], 'td': [re.compile('c\\d+')]}, 'remove_all_attrs': ['div', 'tr'], 'hoist_elements': {('div.d-flex.flex-column.h-100', 'input[title="Grade"]'), ('div.d-flex.flex-column.h-100', 'a.gradeitemheader')}, 'replacements': [('\\n', ''), ('\\s+aria-(?:label|hidden|expanded)="[^"]*"', ''), ('\\s+role\\s*=\\s*(?:"[^"]*"|\\\'[^\\\']*\\\')', ''), ('<div>', ''), ('</div>', ''), ('(<script>)[\\s\\S]*?(</script>)', '\\1M.cfg = {"sesskey":"abcABC123"}\\2')], 'final_selectors': ['head script:not([type])', 'table#user-grades', 'input[name=setmode]']}}

A mapping of Moodle URL patterns to clean_html() kwargs.

def clean_lms_response(response: requests.models.Response, body: str) -> str:
150def clean_lms_response(response: requests.Response, body: str) -> str:
151    """
152    A ResponseModifierFunction that attempt to identify
153    if the requests comes from a Learning Management System (LMS),
154    and clean the response accordingly.
155    """
156
157    # Check the standard LMS Toolkit backend header.
158    raw_backend_type = response.headers.get(lms.model.constants.HEADER_KEY_BACKEND, '').lower()
159
160    if (raw_backend_type == lms.model.constants.BackendType.CANVAS.value):
161        return clean_canvas_response(response, body)
162
163    if (raw_backend_type == lms.model.constants.BackendType.MOODLE.value):
164        return clean_moodle_response(response, body)
165
166    # Try looking inside the header keys.
167    for key in response.headers:
168        key = key.lower().strip()
169
170        if ('blackboard' in key):
171            return clean_blackboard_response(response, body)
172
173        if ('canvas' in key):
174            return clean_canvas_response(response, body)
175
176        if ('moodle' in key):
177            return clean_moodle_response(response, body)
178
179    return body

A ResponseModifierFunction that attempt to identify if the requests comes from a Learning Management System (LMS), and clean the response accordingly.

def clean_blackboard_response(response: requests.models.Response, body: str) -> str:
181def clean_blackboard_response(response: requests.Response, body: str) -> str:
182    """
183    See clean_lms_response(), but specifically for the Blackboard LMS.
184    This function will:
185     - Call _clean_base_response().
186     - Remove specific headers.
187    """
188
189    body = _clean_base_response(response, body)
190
191    # Work on both request and response headers.
192    remove_headers(response, BLACKBOARD_CLEAN_REMOVE_HEADERS)
193
194    # Most blackboard responses are JSON.
195    try:
196        data = edq.util.json.loads(body, strict = True)
197    except Exception:
198        # Response is not JSON.
199        return body
200
201    # Remove any content keys.
202    _recursive_remove_keys(data, set(BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS))
203
204    # Convert body back to a string.
205    body = edq.util.json.dumps(data)
206
207    return body

See clean_lms_response(), but specifically for the Blackboard LMS. This function will:

  • Call _clean_base_response().
  • Remove specific headers.
def clean_canvas_response(response: requests.models.Response, body: str) -> str:
209def clean_canvas_response(response: requests.Response, body: str) -> str:
210    """
211    See clean_lms_response(), but specifically for the Canvas LMS.
212    This function will:
213     - Call _clean_base_response().
214     - Remove content keys: [last_activity_at, total_activity_time]
215    """
216
217    body = _clean_base_response(response, body)
218    url = str(response.request.url)
219
220    if ('/files_api' in url):
221        # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing.
222        parts = urllib.parse.urlsplit(response.headers['location'])
223        parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '')
224        response.headers['location'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG)
225
226    # Most canvas responses are JSON.
227    try:
228        data = edq.util.json.loads(body, strict = True)
229    except Exception:
230        # Response is not JSON.
231        return body
232
233    # Remove any content keys.
234    _recursive_remove_keys(data, set(CANVAS_CLEAN_REMOVE_CONTENT_KEYS))
235
236    # Handle endpoint-specific cases.
237    if ('submissions/update_grades' in url):
238        data.pop('id', None)
239    elif (re.search(r'api/v1/courses/\w+/files', url) is not None):
240        # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing.
241        parts = urllib.parse.urlsplit(data['upload_url'])
242        parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '')
243        data['upload_url'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG)
244
245    # Convert body back to a string.
246    body = edq.util.json.dumps(data)
247
248    return body

See clean_lms_response(), but specifically for the Canvas LMS. This function will:

  • Call _clean_base_response().
  • Remove content keys: [last_activity_at, total_activity_time]
def finalize_canvas_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
250def finalize_canvas_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
251    """ Finalize Canvas exchanges. """
252
253    if (re.search(r'^api/v1/courses/\w+/files$', exchange.url_path) is not None):
254        # Clean out random data for the file upload.
255        data = edq.util.json.loads(str(exchange.response_body))
256
257        body_data = {
258            'upload_params': {
259                'Filename': data['upload_params']['Filename'],
260            },
261            'upload_url': data['upload_url'],
262        }
263
264        exchange.response_body = edq.util.json.dumps(body_data)
265    elif (exchange.url_path == 'files_api'):
266        # File upload calls contain several random pieces of information generated from the server.
267        # Drop information we don't need in testing and make the rest consistent.
268
269        filename = exchange.parameters['Filename']
270        exchange.parameters = {'filename': filename}
271        exchange.files = []
272
273        location = exchange.response_headers.get('location', None)
274        if (location is not None):
275            location = re.sub(r'uuid=.*$', f"uuid={edq.util.hash.sha256_hex(filename)}", str(location))
276
277        exchange.response_headers['location'] = location
278    elif (re.search(r'^api/v1/files/\w+/create_success$', exchange.url_path) is not None):
279        # Change the uuid parameter to match the one set in files_api (which redirects to this URL).
280        data = edq.util.json.loads(str(exchange.response_body))
281        exchange.parameters['uuid'] = edq.util.hash.sha256_hex(data['display_name'])
282        exchange.response_body = edq.util.json.dumps({'id': data['id']})
283
284    return exchange

Finalize Canvas exchanges.

def clean_moodle_response(response: requests.models.Response, body: str) -> str:
286def clean_moodle_response(response: requests.Response, body: str) -> str:
287    """
288    See clean_lms_response(), but specifically for the Moodle LMS.
289    This function will:
290     - Call _clean_base_response().
291    """
292
293    body = _clean_base_response(response, body)
294
295    # Standardize timestamp.
296    current_timestamp_match = re.search(r"boost/theme/(\d{10})/favicon", body)
297    if (current_timestamp_match is not None):
298        body = body.replace(current_timestamp_match.group(1), STANDARDIZED_TIMESTAMP)
299
300    # Standardize session key.
301    session_key_match = re.search(r'"sesskey":"([^"]+)"', body)
302    if (session_key_match is not None):
303        body = body.replace(session_key_match.group(1), STANDARDIZED_SESSION_KEY)
304
305    # Standardize "random" string.
306    random_string_match = re.search(r"'random([a-z0-9]+)'", body)
307    if (random_string_match is not None):
308        body = body.replace(random_string_match.group(1), STANDARDIZED_RANDOM_STRING)
309
310    # Standardize logintoken.
311    logintoken_match = re.search(r'name="logintoken" value="(\w+)"', body)
312    if (logintoken_match is not None):
313        body = body.replace(logintoken_match.group(1), STANDARDIZED_SESSION_KEY)
314
315    # Standardize last access to course.
316    last_access_match = re.search(r'(\d+) secs', body)
317    if (last_access_match is not None):
318        body = body.replace(last_access_match.group(0), f'{STANDARDIZED_TIMESTAMP} secs')
319
320    # Work on both request and response headers.
321    remove_headers(response, MOODLE_CLEAN_REMOVE_HEADERS)
322
323    # Clean HTML responses.
324    for (pattern, clean_params) in MOODLE_HTML_CLEAN.items():
325        if (re.search(pattern, response.url.strip())):
326            body = clean_html(body, **clean_params)
327
328    return body

See clean_lms_response(), but specifically for the Moodle LMS. This function will:

  • Call _clean_base_response().
def clean_html( html: str, filter_elements_by_descendant: Optional[Dict[str, Tuple[str, str]]] = None, delete_elements: Optional[List[str]] = None, attrs_to_keep: Optional[Dict[str, List[str]]] = None, classes_to_keep: Optional[Dict[str, List[Union[str, re.Pattern]]]] = None, remove_all_attrs: Optional[List[str]] = None, hoist_elements: Optional[List[Tuple[str, str]]] = None, replacements: Optional[List[Tuple[str, str]]] = None, final_selectors: Optional[List[str]] = None) -> str:
330def clean_html(
331        html: str,
332        filter_elements_by_descendant: typing.Union[typing.Dict[str, typing.Tuple[str, str]], None] = None,
333        delete_elements: typing.Union[typing.List[str], None] = None,
334        attrs_to_keep: typing.Union[typing.Dict[str, typing.List[str]], None] = None,
335        classes_to_keep: typing.Union[typing.Dict[str, typing.List[typing.Union[str, re.Pattern]]], None] = None,
336        remove_all_attrs: typing.Union[typing.List[str], None] = None,
337        hoist_elements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None,
338        replacements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None,
339        final_selectors: typing.Union[typing.List[str], None] = None,
340        ) -> str:
341    """
342    General purpose HTML cleaning function.
343
344    filter_elements_by_descendant: { selector: (descendant_selector, text), ... }
345    Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text.
346    For example:
347    ```
348    >>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>'
349    >>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')})
350    <div><h2>Hello World!</h2></div>
351    ```
352
353    delete_elements: [ selector, ... ]
354    Deletes any matching elements.
355
356    attrs_to_keep: { selector: [attribute, ...], ... }
357    Keeps only the listed attributes of elements matching the selector.
358
359    classes_to_keep: { selector: [class/pattern, ...], ... }
360    Keeps only listed classes, or classes matching the pattern, of elements matching the selector.
361
362    remove_all_attrs: [ selector, ... ]
363    Removes all attributes of matching elements.
364
365    hoist_elements: [ (parent_selector, child_selector), ... ]
366    For each pair of selectors, the parent element is replaced by the first matching child within the parent.
367    No replacement occurs if both matches are not found.
368
369    replacements: [ (pattern, replacement), ... ]
370    Performs a replacement for all regex matches.
371
372    final_selectors: [ selector, ... ]
373    Replaces html with all matched elements.
374    Useful for targeting only the necessary parts of the response.
375    Defaults to ['body'] if None.
376    """
377
378    if (filter_elements_by_descendant is None):
379        filter_elements_by_descendant = {}
380
381    if (delete_elements is None):
382        delete_elements = []
383
384    if (attrs_to_keep is None):
385        attrs_to_keep = {}
386
387    if (classes_to_keep is None):
388        classes_to_keep = {}
389
390    if (remove_all_attrs is None):
391        remove_all_attrs = []
392
393    if (hoist_elements is None):
394        hoist_elements = []
395
396    if (replacements is None):
397        replacements = [(r'\n', '')]
398
399    if (final_selectors is None):
400        final_selectors = ['body']
401
402    document = bs4.BeautifulSoup(html, 'html.parser')
403
404    # Filter Elements by Descendant
405    for (selector, (descendant_selector, text)) in filter_elements_by_descendant.items():
406        for element in document.select(selector):
407            descendant = element.select_one(descendant_selector)
408            if ((descendant is None) or (descendant.get_text() != text)):
409                element.decompose()
410
411    # Delete Elements
412    for selector in delete_elements:
413        for element in document.select(selector):
414            element.decompose()
415
416    # Keep Attributes
417    for (element_selector, attrs) in attrs_to_keep.items():
418        for element in document.select(element_selector):
419            # Remove extra attributes by keeping only select attributes and replacing the existing attribute dict.
420            element.attrs = {attr: element.attrs[attr] for attr in attrs if (attr in element.attrs)}
421
422    # Keep Classes
423    for (element_selector, keep_classes) in classes_to_keep.items():
424        for element in document.select(element_selector):
425            classes = element.get('class')
426            if ((classes is None) or (len(classes) == 0)):
427                continue
428
429            # Keep only classes listed for this selector.
430            kept = []
431            for keep_class in keep_classes:
432                for class_name in classes:
433                    if (((isinstance(keep_class, re.Pattern)) and (keep_class.fullmatch(class_name))) or (keep_class == class_name)):
434                        kept.append(class_name)
435
436            element['class'] = kept  # type: ignore[assignment]
437
438    # Remove All Attributes
439    for selector in remove_all_attrs:
440        for element in document.select(selector):
441            element.attrs.clear()
442
443    # Element Hoisting
444    for (parent_selector, child_selector) in hoist_elements:
445        for parent in document.select(parent_selector):
446            child = parent.select_one(child_selector)
447            if (child is None):
448                continue
449
450            parent.replace_with(child.extract())
451
452    # Final Selectors
453    elements = []
454    for final_selector in final_selectors:
455        for element in document.select(final_selector):
456            elements.append(element)
457
458    document_string = "".join([str(element) for element in elements])
459
460    # Replacements
461    for (pattern, replacement) in replacements:
462        document_string = re.sub(pattern, replacement, document_string)
463
464    return document_string

General purpose HTML cleaning function.

filter_elements_by_descendant: { selector: (descendant_selector, text), ... } Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text. For example:

>>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>'
>>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')})
<div><h2>Hello World!</h2></div>

delete_elements: [ selector, ... ] Deletes any matching elements.

attrs_to_keep: { selector: [attribute, ...], ... } Keeps only the listed attributes of elements matching the selector.

classes_to_keep: { selector: [class/pattern, ...], ... } Keeps only listed classes, or classes matching the pattern, of elements matching the selector.

remove_all_attrs: [ selector, ... ] Removes all attributes of matching elements.

hoist_elements: [ (parent_selector, child_selector), ... ] For each pair of selectors, the parent element is replaced by the first matching child within the parent. No replacement occurs if both matches are not found.

replacements: [ (pattern, replacement), ... ] Performs a replacement for all regex matches.

final_selectors: [ selector, ... ] Replaces html with all matched elements. Useful for targeting only the necessary parts of the response. Defaults to ['body'] if None.

def remove_headers(response: requests.models.Response, headers_to_remove: Set[str]) -> None:
466def remove_headers(response: requests.Response, headers_to_remove: typing.Set[str]) -> None:
467    """
468    Remove headers from response and response's request.
469    """
470
471    for headers in [response.headers, response.request.headers]:
472        for key in list(headers.keys()):  # type: ignore[attr-defined]
473            if (key.strip().lower() in headers_to_remove):
474                headers.pop(key, None)  # type: ignore[attr-defined]

Remove headers from response and response's request.

def finalize_moodle_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
476def finalize_moodle_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange:
477    """ Finalize Moodle exchanges. """
478
479    for param in MOODLE_FINALIZE_REMOVE_PARAMS:
480        exchange.parameters.pop(param, None)
481
482    # Standardize session key.
483    if ('sesskey' in exchange.parameters):
484        exchange.parameters['sesskey'] = STANDARDIZED_SESSION_KEY
485
486    return exchange

Finalize Moodle exchanges.

def parse_cookies( text_cookies: Optional[str], strip_key_prefix: bool = True) -> Dict[str, Any]:
527def parse_cookies(
528        text_cookies: typing.Union[str, None],
529        strip_key_prefix: bool = True,
530        ) -> typing.Dict[str, typing.Any]:
531    """ Parse cookies out of a text string. """
532
533    cookies: typing.Dict[str, typing.Any] = {}
534
535    if (text_cookies is None):
536        return cookies
537
538    text_cookies = text_cookies.strip()
539    if (len(text_cookies) == 0):
540        return cookies
541
542    for cookie in text_cookies.split('; '):
543        parts = cookie.split('=', maxsplit = 1)
544
545        key = parts[0].lower()
546
547        if (strip_key_prefix):
548            key = key.split(', ')[-1]
549
550        if (len(parts) == 1):
551            cookies[key] = True
552        else:
553            cookies[key] = parts[1]
554
555    return cookies

Parse cookies out of a text string.