quizcomp.parser.document
1import os 2import typing 3 4import edq.util.dirent 5import edq.util.json 6import edq.util.serial 7import markdown_it.token 8 9import quizcomp.model.constants 10import quizcomp.parser.ast 11import quizcomp.parser.render 12import quizcomp.parser.common 13 14class ParsedDocument(edq.util.serial.PODSerializer): 15 """ 16 The result of parsing some text. 17 18 Users that modify the tokens of a document directly should call tokens_updated() after. 19 """ 20 21 def __init__(self, 22 text: typing.Union[str, None] = None, 23 tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None, 24 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 25 ) -> None: 26 if (text is None): 27 text = '' 28 29 if (tokens is None): 30 tokens = [] 31 32 if ((len(text) == 0) and (len(tokens) > 0)): 33 raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.") 34 35 if ((len(text) > 0) and (len(tokens) == 0)): 36 text, tokens = quizcomp.parser.render._parse_text(text) 37 38 self.text: str = text 39 """ The cleaned text that was parsed to create this document. """ 40 41 self._tokens: typing.List[markdown_it.token.Token] = tokens 42 """ The tokens that were parsed from the starting text. """ 43 44 if (context is None): 45 context = edq.util.serial.SerializationContext() 46 47 self.context: edq.util.serial.SerializationContext = context 48 """ The context the text was parsed in. """ 49 50 def tokens_updated(self) -> None: 51 """ Update the cached text in the document under the assumption that tokens were modified. """ 52 53 self.text = self.to_md() 54 55 def to_canvas(self, **kwargs: typing.Any) -> str: 56 """ Render this document to Canvas-specific HTML. """ 57 58 return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs) 59 60 def to_md(self, **kwargs: typing.Any) -> str: 61 """ Render this document to Markdown. """ 62 63 return self._render(quizcomp.model.constants.Format.MD, **kwargs) 64 65 def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str: 66 """ Render this document to JSON. """ 67 68 data = { 69 'text': self.text, 70 'ast': self.get_ast().to_pod(), 71 } 72 return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys) 73 74 def to_tex(self, **kwargs: typing.Any) -> str: 75 """ Render this document to TeX. """ 76 77 return self._render(quizcomp.model.constants.Format.TEX, **kwargs) 78 79 def to_text(self, **kwargs: typing.Any) -> str: 80 """ Render this document to simple text. """ 81 82 return self._render(quizcomp.model.constants.Format.TEXT, **kwargs) 83 84 def to_html(self, **kwargs: typing.Any) -> str: 85 """ Render this document to HTML. """ 86 87 return self._render(quizcomp.model.constants.Format.HTML, **kwargs) 88 89 def _render(self, 90 format: quizcomp.model.constants.Format, 91 context: typing.Union[quizcomp.parser.common.RenderContext, None] = None, 92 **kwargs: typing.Any) -> str: 93 """ Render this document to the specified format. """ 94 95 if (context is None): 96 context = quizcomp.parser.common.RenderContext(**kwargs) 97 98 if (kwargs.get('base_dir', None) is None): 99 context.base_dir = self.context.base_dir 100 101 if (context.source_path is None): 102 context.source_path = self.context.source_path 103 104 env = { 105 quizcomp.parser.common.ENV_KEY_CONTEXT: context, 106 } 107 108 return quizcomp.parser.render.render(format, self._tokens, env = env, **kwargs) 109 110 def collect_placeholders(self) -> typing.Set[str]: 111 """ 112 Fetch all the answer placeholders in this document. 113 """ 114 115 tokens = self._collect_tokens(self._tokens, 'placeholder') 116 return {token.content for token in tokens} 117 118 def collect_images(self) -> typing.List[markdown_it.token.Token]: 119 """ 120 Fetch all the image tokens in this document. 121 """ 122 123 return self._collect_tokens(self._tokens, 'image') 124 125 def _collect_tokens(self, 126 tokens: typing.Union[typing.List[markdown_it.token.Token], None], 127 token_type: str, 128 ) -> typing.List[markdown_it.token.Token]: 129 """ Recursively collect tokens of the given type. """ 130 131 contents: typing.List[markdown_it.token.Token] = [] 132 133 if ((tokens is None) or (len(tokens) == 0)): 134 return contents 135 136 for token in tokens: 137 if (token.type == token_type): 138 contents.append(token) 139 140 contents += self._collect_tokens(token.children, token_type) 141 142 return contents 143 144 def is_empty(self) -> bool: 145 """ Check if this document contains any content (tokens). """ 146 147 return (len(self._tokens) == 0) 148 149 def _serialization_is_empty(self) -> bool: 150 """ A special method for the serialization library to check. """ 151 152 return self.is_empty() 153 154 def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str: 155 """ Convert this document to the specified format. """ 156 157 formatter = getattr(self, 'to_' + format.value) 158 if (formatter is None): 159 raise ValueError(f"Unknown format '{format.value}'.") 160 161 return str(formatter(**kwargs)) 162 163 def get_ast(self) -> quizcomp.parser.ast.ASTNode: 164 """ 165 Get a representation of this document's AST. 166 """ 167 168 return quizcomp.parser.ast.build(self._tokens) 169 170 def __repr__(self) -> str: 171 return self.text 172 173 def to_pod(self, 174 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 175 ) -> str: 176 return self.text 177 178 @classmethod 179 def parse_text(cls, 180 text: typing.Union[str, 'ParsedDocument'], 181 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 182 ) -> 'ParsedDocument': 183 """ 184 Parse some text into a document. 185 If the text is already a document, that same document will be returned. 186 """ 187 188 if (isinstance(text, ParsedDocument)): 189 return text 190 191 if (context is None): 192 context = edq.util.serial.SerializationContext() 193 194 text, tokens = quizcomp.parser.render._parse_text(text) 195 return quizcomp.parser.document.ParsedDocument(text, tokens, context) 196 197 @classmethod 198 def parse_file(cls, 199 raw_path: str, 200 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 201 ) -> 'ParsedDocument': 202 """ 203 Parse a text file into a document. 204 205 If a context is provided, 206 a copy will be made with the base dir and source path updated. 207 """ 208 209 if (context is None): 210 context = edq.util.serial.SerializationContext() 211 else: 212 context = context.copy() 213 214 if (context.base_dir is None): 215 context.base_dir = os.path.dirname(os.path.abspath(raw_path)) 216 217 # Prepend the base dir if the path is not absolute. 218 if (not os.path.isabs(raw_path)): 219 raw_path = os.path.join(context.base_dir, raw_path) 220 221 context.source_path = os.path.abspath(raw_path) 222 context.base_dir = os.path.dirname(context.source_path) 223 224 if (not os.path.isfile(context.source_path)): 225 raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.") 226 227 text = edq.util.dirent.read_file(context.source_path) 228 229 return cls.parse_text(text, context)
15class ParsedDocument(edq.util.serial.PODSerializer): 16 """ 17 The result of parsing some text. 18 19 Users that modify the tokens of a document directly should call tokens_updated() after. 20 """ 21 22 def __init__(self, 23 text: typing.Union[str, None] = None, 24 tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None, 25 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 26 ) -> None: 27 if (text is None): 28 text = '' 29 30 if (tokens is None): 31 tokens = [] 32 33 if ((len(text) == 0) and (len(tokens) > 0)): 34 raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.") 35 36 if ((len(text) > 0) and (len(tokens) == 0)): 37 text, tokens = quizcomp.parser.render._parse_text(text) 38 39 self.text: str = text 40 """ The cleaned text that was parsed to create this document. """ 41 42 self._tokens: typing.List[markdown_it.token.Token] = tokens 43 """ The tokens that were parsed from the starting text. """ 44 45 if (context is None): 46 context = edq.util.serial.SerializationContext() 47 48 self.context: edq.util.serial.SerializationContext = context 49 """ The context the text was parsed in. """ 50 51 def tokens_updated(self) -> None: 52 """ Update the cached text in the document under the assumption that tokens were modified. """ 53 54 self.text = self.to_md() 55 56 def to_canvas(self, **kwargs: typing.Any) -> str: 57 """ Render this document to Canvas-specific HTML. """ 58 59 return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs) 60 61 def to_md(self, **kwargs: typing.Any) -> str: 62 """ Render this document to Markdown. """ 63 64 return self._render(quizcomp.model.constants.Format.MD, **kwargs) 65 66 def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str: 67 """ Render this document to JSON. """ 68 69 data = { 70 'text': self.text, 71 'ast': self.get_ast().to_pod(), 72 } 73 return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys) 74 75 def to_tex(self, **kwargs: typing.Any) -> str: 76 """ Render this document to TeX. """ 77 78 return self._render(quizcomp.model.constants.Format.TEX, **kwargs) 79 80 def to_text(self, **kwargs: typing.Any) -> str: 81 """ Render this document to simple text. """ 82 83 return self._render(quizcomp.model.constants.Format.TEXT, **kwargs) 84 85 def to_html(self, **kwargs: typing.Any) -> str: 86 """ Render this document to HTML. """ 87 88 return self._render(quizcomp.model.constants.Format.HTML, **kwargs) 89 90 def _render(self, 91 format: quizcomp.model.constants.Format, 92 context: typing.Union[quizcomp.parser.common.RenderContext, None] = None, 93 **kwargs: typing.Any) -> str: 94 """ Render this document to the specified format. """ 95 96 if (context is None): 97 context = quizcomp.parser.common.RenderContext(**kwargs) 98 99 if (kwargs.get('base_dir', None) is None): 100 context.base_dir = self.context.base_dir 101 102 if (context.source_path is None): 103 context.source_path = self.context.source_path 104 105 env = { 106 quizcomp.parser.common.ENV_KEY_CONTEXT: context, 107 } 108 109 return quizcomp.parser.render.render(format, self._tokens, env = env, **kwargs) 110 111 def collect_placeholders(self) -> typing.Set[str]: 112 """ 113 Fetch all the answer placeholders in this document. 114 """ 115 116 tokens = self._collect_tokens(self._tokens, 'placeholder') 117 return {token.content for token in tokens} 118 119 def collect_images(self) -> typing.List[markdown_it.token.Token]: 120 """ 121 Fetch all the image tokens in this document. 122 """ 123 124 return self._collect_tokens(self._tokens, 'image') 125 126 def _collect_tokens(self, 127 tokens: typing.Union[typing.List[markdown_it.token.Token], None], 128 token_type: str, 129 ) -> typing.List[markdown_it.token.Token]: 130 """ Recursively collect tokens of the given type. """ 131 132 contents: typing.List[markdown_it.token.Token] = [] 133 134 if ((tokens is None) or (len(tokens) == 0)): 135 return contents 136 137 for token in tokens: 138 if (token.type == token_type): 139 contents.append(token) 140 141 contents += self._collect_tokens(token.children, token_type) 142 143 return contents 144 145 def is_empty(self) -> bool: 146 """ Check if this document contains any content (tokens). """ 147 148 return (len(self._tokens) == 0) 149 150 def _serialization_is_empty(self) -> bool: 151 """ A special method for the serialization library to check. """ 152 153 return self.is_empty() 154 155 def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str: 156 """ Convert this document to the specified format. """ 157 158 formatter = getattr(self, 'to_' + format.value) 159 if (formatter is None): 160 raise ValueError(f"Unknown format '{format.value}'.") 161 162 return str(formatter(**kwargs)) 163 164 def get_ast(self) -> quizcomp.parser.ast.ASTNode: 165 """ 166 Get a representation of this document's AST. 167 """ 168 169 return quizcomp.parser.ast.build(self._tokens) 170 171 def __repr__(self) -> str: 172 return self.text 173 174 def to_pod(self, 175 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 176 ) -> str: 177 return self.text 178 179 @classmethod 180 def parse_text(cls, 181 text: typing.Union[str, 'ParsedDocument'], 182 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 183 ) -> 'ParsedDocument': 184 """ 185 Parse some text into a document. 186 If the text is already a document, that same document will be returned. 187 """ 188 189 if (isinstance(text, ParsedDocument)): 190 return text 191 192 if (context is None): 193 context = edq.util.serial.SerializationContext() 194 195 text, tokens = quizcomp.parser.render._parse_text(text) 196 return quizcomp.parser.document.ParsedDocument(text, tokens, context) 197 198 @classmethod 199 def parse_file(cls, 200 raw_path: str, 201 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 202 ) -> 'ParsedDocument': 203 """ 204 Parse a text file into a document. 205 206 If a context is provided, 207 a copy will be made with the base dir and source path updated. 208 """ 209 210 if (context is None): 211 context = edq.util.serial.SerializationContext() 212 else: 213 context = context.copy() 214 215 if (context.base_dir is None): 216 context.base_dir = os.path.dirname(os.path.abspath(raw_path)) 217 218 # Prepend the base dir if the path is not absolute. 219 if (not os.path.isabs(raw_path)): 220 raw_path = os.path.join(context.base_dir, raw_path) 221 222 context.source_path = os.path.abspath(raw_path) 223 context.base_dir = os.path.dirname(context.source_path) 224 225 if (not os.path.isfile(context.source_path)): 226 raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.") 227 228 text = edq.util.dirent.read_file(context.source_path) 229 230 return cls.parse_text(text, context)
The result of parsing some text.
Users that modify the tokens of a document directly should call tokens_updated() after.
22 def __init__(self, 23 text: typing.Union[str, None] = None, 24 tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None, 25 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 26 ) -> None: 27 if (text is None): 28 text = '' 29 30 if (tokens is None): 31 tokens = [] 32 33 if ((len(text) == 0) and (len(tokens) > 0)): 34 raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.") 35 36 if ((len(text) > 0) and (len(tokens) == 0)): 37 text, tokens = quizcomp.parser.render._parse_text(text) 38 39 self.text: str = text 40 """ The cleaned text that was parsed to create this document. """ 41 42 self._tokens: typing.List[markdown_it.token.Token] = tokens 43 """ The tokens that were parsed from the starting text. """ 44 45 if (context is None): 46 context = edq.util.serial.SerializationContext() 47 48 self.context: edq.util.serial.SerializationContext = context 49 """ The context the text was parsed in. """
51 def tokens_updated(self) -> None: 52 """ Update the cached text in the document under the assumption that tokens were modified. """ 53 54 self.text = self.to_md()
Update the cached text in the document under the assumption that tokens were modified.
56 def to_canvas(self, **kwargs: typing.Any) -> str: 57 """ Render this document to Canvas-specific HTML. """ 58 59 return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs)
Render this document to Canvas-specific HTML.
61 def to_md(self, **kwargs: typing.Any) -> str: 62 """ Render this document to Markdown. """ 63 64 return self._render(quizcomp.model.constants.Format.MD, **kwargs)
Render this document to Markdown.
66 def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str: 67 """ Render this document to JSON. """ 68 69 data = { 70 'text': self.text, 71 'ast': self.get_ast().to_pod(), 72 } 73 return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys)
Render this document to JSON.
75 def to_tex(self, **kwargs: typing.Any) -> str: 76 """ Render this document to TeX. """ 77 78 return self._render(quizcomp.model.constants.Format.TEX, **kwargs)
Render this document to TeX.
80 def to_text(self, **kwargs: typing.Any) -> str: 81 """ Render this document to simple text. """ 82 83 return self._render(quizcomp.model.constants.Format.TEXT, **kwargs)
Render this document to simple text.
85 def to_html(self, **kwargs: typing.Any) -> str: 86 """ Render this document to HTML. """ 87 88 return self._render(quizcomp.model.constants.Format.HTML, **kwargs)
Render this document to HTML.
111 def collect_placeholders(self) -> typing.Set[str]: 112 """ 113 Fetch all the answer placeholders in this document. 114 """ 115 116 tokens = self._collect_tokens(self._tokens, 'placeholder') 117 return {token.content for token in tokens}
Fetch all the answer placeholders in this document.
119 def collect_images(self) -> typing.List[markdown_it.token.Token]: 120 """ 121 Fetch all the image tokens in this document. 122 """ 123 124 return self._collect_tokens(self._tokens, 'image')
Fetch all the image tokens in this document.
145 def is_empty(self) -> bool: 146 """ Check if this document contains any content (tokens). """ 147 148 return (len(self._tokens) == 0)
Check if this document contains any content (tokens).
155 def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str: 156 """ Convert this document to the specified format. """ 157 158 formatter = getattr(self, 'to_' + format.value) 159 if (formatter is None): 160 raise ValueError(f"Unknown format '{format.value}'.") 161 162 return str(formatter(**kwargs))
Convert this document to the specified format.
164 def get_ast(self) -> quizcomp.parser.ast.ASTNode: 165 """ 166 Get a representation of this document's AST. 167 """ 168 169 return quizcomp.parser.ast.build(self._tokens)
Get a representation of this document's AST.
174 def to_pod(self, 175 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 176 ) -> str: 177 return self.text
Get a POD representation of this object.
The default implementation will convert to a dict (similar to a DictSerializer).
179 @classmethod 180 def parse_text(cls, 181 text: typing.Union[str, 'ParsedDocument'], 182 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 183 ) -> 'ParsedDocument': 184 """ 185 Parse some text into a document. 186 If the text is already a document, that same document will be returned. 187 """ 188 189 if (isinstance(text, ParsedDocument)): 190 return text 191 192 if (context is None): 193 context = edq.util.serial.SerializationContext() 194 195 text, tokens = quizcomp.parser.render._parse_text(text) 196 return quizcomp.parser.document.ParsedDocument(text, tokens, context)
Parse some text into a document. If the text is already a document, that same document will be returned.
198 @classmethod 199 def parse_file(cls, 200 raw_path: str, 201 context: typing.Union[edq.util.serial.SerializationContext, None] = None, 202 ) -> 'ParsedDocument': 203 """ 204 Parse a text file into a document. 205 206 If a context is provided, 207 a copy will be made with the base dir and source path updated. 208 """ 209 210 if (context is None): 211 context = edq.util.serial.SerializationContext() 212 else: 213 context = context.copy() 214 215 if (context.base_dir is None): 216 context.base_dir = os.path.dirname(os.path.abspath(raw_path)) 217 218 # Prepend the base dir if the path is not absolute. 219 if (not os.path.isabs(raw_path)): 220 raw_path = os.path.join(context.base_dir, raw_path) 221 222 context.source_path = os.path.abspath(raw_path) 223 context.base_dir = os.path.dirname(context.source_path) 224 225 if (not os.path.isfile(context.source_path)): 226 raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.") 227 228 text = edq.util.dirent.read_file(context.source_path) 229 230 return cls.parse_text(text, context)
Parse a text file into a document.
If a context is provided, a copy will be made with the base dir and source path updated.