quizcomp.parser.document

  1import os
  2import typing
  3
  4import edq.util.dirent
  5import edq.util.json
  6import edq.util.serial
  7import markdown_it.token
  8
  9import quizcomp.model.constants
 10import quizcomp.parser.ast
 11import quizcomp.parser.render
 12import quizcomp.parser.common
 13
 14class ParsedDocument(edq.util.serial.PODSerializer):
 15    """
 16    The result of parsing some text.
 17
 18    Users that modify the tokens of a document directly should call tokens_updated() after.
 19    """
 20
 21    def __init__(self,
 22            text: typing.Union[str, None] = None,
 23            tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None,
 24            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
 25            ) -> None:
 26        if (text is None):
 27            text = ''
 28
 29        if (tokens is None):
 30            tokens = []
 31
 32        if ((len(text) == 0) and (len(tokens) > 0)):
 33            raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.")
 34
 35        if ((len(text) > 0) and (len(tokens) == 0)):
 36            text, tokens = quizcomp.parser.render._parse_text(text)
 37
 38        self.text: str = text
 39        """ The cleaned text that was parsed to create this document. """
 40
 41        self._tokens: typing.List[markdown_it.token.Token] = tokens
 42        """ The tokens that were parsed from the starting text. """
 43
 44        if (context is None):
 45            context = edq.util.serial.SerializationContext()
 46
 47        self.context: edq.util.serial.SerializationContext = context
 48        """ The context the text was parsed in. """
 49
 50    def tokens_updated(self) -> None:
 51        """ Update the cached text in the document under the assumption that tokens were modified. """
 52
 53        self.text = self.to_md()
 54
 55    def to_canvas(self, **kwargs: typing.Any) -> str:
 56        """ Render this document to Canvas-specific HTML. """
 57
 58        return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs)
 59
 60    def to_md(self, **kwargs: typing.Any) -> str:
 61        """ Render this document to Markdown. """
 62
 63        return self._render(quizcomp.model.constants.Format.MD, **kwargs)
 64
 65    def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str:
 66        """ Render this document to JSON. """
 67
 68        data = {
 69            'text': self.text,
 70            'ast': self.get_ast().to_pod(),
 71        }
 72        return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys)
 73
 74    def to_tex(self, **kwargs: typing.Any) -> str:
 75        """ Render this document to TeX. """
 76
 77        return self._render(quizcomp.model.constants.Format.TEX, **kwargs)
 78
 79    def to_text(self, **kwargs: typing.Any) -> str:
 80        """ Render this document to simple text. """
 81
 82        return self._render(quizcomp.model.constants.Format.TEXT, **kwargs)
 83
 84    def to_html(self, **kwargs: typing.Any) -> str:
 85        """ Render this document to HTML. """
 86
 87        return self._render(quizcomp.model.constants.Format.HTML, **kwargs)
 88
 89    def _render(self,
 90            format: quizcomp.model.constants.Format,
 91            context: typing.Union[quizcomp.parser.common.RenderContext, None] = None,
 92            **kwargs: typing.Any) -> str:
 93        """ Render this document to the specified format. """
 94
 95        if (context is None):
 96            context = quizcomp.parser.common.RenderContext(**kwargs)
 97
 98            if (kwargs.get('base_dir', None) is None):
 99                context.base_dir = self.context.base_dir
100
101            if (context.source_path is None):
102                context.source_path = self.context.source_path
103
104        env = {
105            quizcomp.parser.common.ENV_KEY_CONTEXT: context,
106        }
107
108        return quizcomp.parser.render.render(format, self._tokens, env = env, **kwargs)
109
110    def collect_placeholders(self) -> typing.Set[str]:
111        """
112        Fetch all the answer placeholders in this document.
113        """
114
115        tokens = self._collect_tokens(self._tokens, 'placeholder')
116        return {token.content for token in tokens}
117
118    def collect_images(self) -> typing.List[markdown_it.token.Token]:
119        """
120        Fetch all the image tokens in this document.
121        """
122
123        return self._collect_tokens(self._tokens, 'image')
124
125    def _collect_tokens(self,
126            tokens: typing.Union[typing.List[markdown_it.token.Token], None],
127            token_type: str,
128            ) -> typing.List[markdown_it.token.Token]:
129        """ Recursively collect tokens of the given type. """
130
131        contents: typing.List[markdown_it.token.Token] = []
132
133        if ((tokens is None) or (len(tokens) == 0)):
134            return contents
135
136        for token in tokens:
137            if (token.type == token_type):
138                contents.append(token)
139
140            contents += self._collect_tokens(token.children, token_type)
141
142        return contents
143
144    def is_empty(self) -> bool:
145        """ Check if this document contains any content (tokens). """
146
147        return (len(self._tokens) == 0)
148
149    def _serialization_is_empty(self) -> bool:
150        """ A special method for the serialization library to check. """
151
152        return self.is_empty()
153
154    def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str:
155        """ Convert this document to the specified format. """
156
157        formatter = getattr(self, 'to_' + format.value)
158        if (formatter is None):
159            raise ValueError(f"Unknown format '{format.value}'.")
160
161        return str(formatter(**kwargs))
162
163    def get_ast(self) -> quizcomp.parser.ast.ASTNode:
164        """
165        Get a representation of this document's AST.
166        """
167
168        return quizcomp.parser.ast.build(self._tokens)
169
170    def __repr__(self) -> str:
171        return self.text
172
173    def to_pod(self,
174            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
175            ) -> str:
176        return self.text
177
178    @classmethod
179    def parse_text(cls,
180            text: typing.Union[str, 'ParsedDocument'],
181            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
182            ) -> 'ParsedDocument':
183        """
184        Parse some text into a document.
185        If the text is already a document, that same document will be returned.
186        """
187
188        if (isinstance(text, ParsedDocument)):
189            return text
190
191        if (context is None):
192            context = edq.util.serial.SerializationContext()
193
194        text, tokens = quizcomp.parser.render._parse_text(text)
195        return quizcomp.parser.document.ParsedDocument(text, tokens, context)
196
197    @classmethod
198    def parse_file(cls,
199            raw_path: str,
200            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
201            ) -> 'ParsedDocument':
202        """
203        Parse a text file into a document.
204
205        If a context is provided,
206        a copy will be made with the base dir and source path updated.
207        """
208
209        if (context is None):
210            context = edq.util.serial.SerializationContext()
211        else:
212            context = context.copy()
213
214        if (context.base_dir is None):
215            context.base_dir = os.path.dirname(os.path.abspath(raw_path))
216
217        # Prepend the base dir if the path is not absolute.
218        if (not os.path.isabs(raw_path)):
219            raw_path = os.path.join(context.base_dir, raw_path)
220
221        context.source_path = os.path.abspath(raw_path)
222        context.base_dir = os.path.dirname(context.source_path)
223
224        if (not os.path.isfile(context.source_path)):
225            raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.")
226
227        text = edq.util.dirent.read_file(context.source_path)
228
229        return cls.parse_text(text, context)
class ParsedDocument(edq.util.serial.PODSerializer):
 15class ParsedDocument(edq.util.serial.PODSerializer):
 16    """
 17    The result of parsing some text.
 18
 19    Users that modify the tokens of a document directly should call tokens_updated() after.
 20    """
 21
 22    def __init__(self,
 23            text: typing.Union[str, None] = None,
 24            tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None,
 25            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
 26            ) -> None:
 27        if (text is None):
 28            text = ''
 29
 30        if (tokens is None):
 31            tokens = []
 32
 33        if ((len(text) == 0) and (len(tokens) > 0)):
 34            raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.")
 35
 36        if ((len(text) > 0) and (len(tokens) == 0)):
 37            text, tokens = quizcomp.parser.render._parse_text(text)
 38
 39        self.text: str = text
 40        """ The cleaned text that was parsed to create this document. """
 41
 42        self._tokens: typing.List[markdown_it.token.Token] = tokens
 43        """ The tokens that were parsed from the starting text. """
 44
 45        if (context is None):
 46            context = edq.util.serial.SerializationContext()
 47
 48        self.context: edq.util.serial.SerializationContext = context
 49        """ The context the text was parsed in. """
 50
 51    def tokens_updated(self) -> None:
 52        """ Update the cached text in the document under the assumption that tokens were modified. """
 53
 54        self.text = self.to_md()
 55
 56    def to_canvas(self, **kwargs: typing.Any) -> str:
 57        """ Render this document to Canvas-specific HTML. """
 58
 59        return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs)
 60
 61    def to_md(self, **kwargs: typing.Any) -> str:
 62        """ Render this document to Markdown. """
 63
 64        return self._render(quizcomp.model.constants.Format.MD, **kwargs)
 65
 66    def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str:
 67        """ Render this document to JSON. """
 68
 69        data = {
 70            'text': self.text,
 71            'ast': self.get_ast().to_pod(),
 72        }
 73        return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys)
 74
 75    def to_tex(self, **kwargs: typing.Any) -> str:
 76        """ Render this document to TeX. """
 77
 78        return self._render(quizcomp.model.constants.Format.TEX, **kwargs)
 79
 80    def to_text(self, **kwargs: typing.Any) -> str:
 81        """ Render this document to simple text. """
 82
 83        return self._render(quizcomp.model.constants.Format.TEXT, **kwargs)
 84
 85    def to_html(self, **kwargs: typing.Any) -> str:
 86        """ Render this document to HTML. """
 87
 88        return self._render(quizcomp.model.constants.Format.HTML, **kwargs)
 89
 90    def _render(self,
 91            format: quizcomp.model.constants.Format,
 92            context: typing.Union[quizcomp.parser.common.RenderContext, None] = None,
 93            **kwargs: typing.Any) -> str:
 94        """ Render this document to the specified format. """
 95
 96        if (context is None):
 97            context = quizcomp.parser.common.RenderContext(**kwargs)
 98
 99            if (kwargs.get('base_dir', None) is None):
100                context.base_dir = self.context.base_dir
101
102            if (context.source_path is None):
103                context.source_path = self.context.source_path
104
105        env = {
106            quizcomp.parser.common.ENV_KEY_CONTEXT: context,
107        }
108
109        return quizcomp.parser.render.render(format, self._tokens, env = env, **kwargs)
110
111    def collect_placeholders(self) -> typing.Set[str]:
112        """
113        Fetch all the answer placeholders in this document.
114        """
115
116        tokens = self._collect_tokens(self._tokens, 'placeholder')
117        return {token.content for token in tokens}
118
119    def collect_images(self) -> typing.List[markdown_it.token.Token]:
120        """
121        Fetch all the image tokens in this document.
122        """
123
124        return self._collect_tokens(self._tokens, 'image')
125
126    def _collect_tokens(self,
127            tokens: typing.Union[typing.List[markdown_it.token.Token], None],
128            token_type: str,
129            ) -> typing.List[markdown_it.token.Token]:
130        """ Recursively collect tokens of the given type. """
131
132        contents: typing.List[markdown_it.token.Token] = []
133
134        if ((tokens is None) or (len(tokens) == 0)):
135            return contents
136
137        for token in tokens:
138            if (token.type == token_type):
139                contents.append(token)
140
141            contents += self._collect_tokens(token.children, token_type)
142
143        return contents
144
145    def is_empty(self) -> bool:
146        """ Check if this document contains any content (tokens). """
147
148        return (len(self._tokens) == 0)
149
150    def _serialization_is_empty(self) -> bool:
151        """ A special method for the serialization library to check. """
152
153        return self.is_empty()
154
155    def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str:
156        """ Convert this document to the specified format. """
157
158        formatter = getattr(self, 'to_' + format.value)
159        if (formatter is None):
160            raise ValueError(f"Unknown format '{format.value}'.")
161
162        return str(formatter(**kwargs))
163
164    def get_ast(self) -> quizcomp.parser.ast.ASTNode:
165        """
166        Get a representation of this document's AST.
167        """
168
169        return quizcomp.parser.ast.build(self._tokens)
170
171    def __repr__(self) -> str:
172        return self.text
173
174    def to_pod(self,
175            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
176            ) -> str:
177        return self.text
178
179    @classmethod
180    def parse_text(cls,
181            text: typing.Union[str, 'ParsedDocument'],
182            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
183            ) -> 'ParsedDocument':
184        """
185        Parse some text into a document.
186        If the text is already a document, that same document will be returned.
187        """
188
189        if (isinstance(text, ParsedDocument)):
190            return text
191
192        if (context is None):
193            context = edq.util.serial.SerializationContext()
194
195        text, tokens = quizcomp.parser.render._parse_text(text)
196        return quizcomp.parser.document.ParsedDocument(text, tokens, context)
197
198    @classmethod
199    def parse_file(cls,
200            raw_path: str,
201            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
202            ) -> 'ParsedDocument':
203        """
204        Parse a text file into a document.
205
206        If a context is provided,
207        a copy will be made with the base dir and source path updated.
208        """
209
210        if (context is None):
211            context = edq.util.serial.SerializationContext()
212        else:
213            context = context.copy()
214
215        if (context.base_dir is None):
216            context.base_dir = os.path.dirname(os.path.abspath(raw_path))
217
218        # Prepend the base dir if the path is not absolute.
219        if (not os.path.isabs(raw_path)):
220            raw_path = os.path.join(context.base_dir, raw_path)
221
222        context.source_path = os.path.abspath(raw_path)
223        context.base_dir = os.path.dirname(context.source_path)
224
225        if (not os.path.isfile(context.source_path)):
226            raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.")
227
228        text = edq.util.dirent.read_file(context.source_path)
229
230        return cls.parse_text(text, context)

The result of parsing some text.

Users that modify the tokens of a document directly should call tokens_updated() after.

ParsedDocument( text: Optional[str] = None, tokens: Optional[List[markdown_it.token.Token]] = None, context: Optional[edq.util.common.SerializationContext] = None)
22    def __init__(self,
23            text: typing.Union[str, None] = None,
24            tokens: typing.Union[typing.List[markdown_it.token.Token], None] = None,
25            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
26            ) -> None:
27        if (text is None):
28            text = ''
29
30        if (tokens is None):
31            tokens = []
32
33        if ((len(text) == 0) and (len(tokens) > 0)):
34            raise ValueError(f"Cannot create a document that contains tokens, but not text: {tokens}.")
35
36        if ((len(text) > 0) and (len(tokens) == 0)):
37            text, tokens = quizcomp.parser.render._parse_text(text)
38
39        self.text: str = text
40        """ The cleaned text that was parsed to create this document. """
41
42        self._tokens: typing.List[markdown_it.token.Token] = tokens
43        """ The tokens that were parsed from the starting text. """
44
45        if (context is None):
46            context = edq.util.serial.SerializationContext()
47
48        self.context: edq.util.serial.SerializationContext = context
49        """ The context the text was parsed in. """
text: str

The cleaned text that was parsed to create this document.

context: edq.util.common.SerializationContext

The context the text was parsed in.

def tokens_updated(self) -> None:
51    def tokens_updated(self) -> None:
52        """ Update the cached text in the document under the assumption that tokens were modified. """
53
54        self.text = self.to_md()

Update the cached text in the document under the assumption that tokens were modified.

def to_canvas(self, **kwargs: Any) -> str:
56    def to_canvas(self, **kwargs: typing.Any) -> str:
57        """ Render this document to Canvas-specific HTML. """
58
59        return self._render(quizcomp.model.constants.Format.CANVAS, **kwargs)

Render this document to Canvas-specific HTML.

def to_md(self, **kwargs: Any) -> str:
61    def to_md(self, **kwargs: typing.Any) -> str:
62        """ Render this document to Markdown. """
63
64        return self._render(quizcomp.model.constants.Format.MD, **kwargs)

Render this document to Markdown.

def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: Any) -> str:
66    def to_json(self, indent: int = 4, sort_keys: bool = True, **kwargs: typing.Any) -> str:
67        """ Render this document to JSON. """
68
69        data = {
70            'text': self.text,
71            'ast': self.get_ast().to_pod(),
72        }
73        return edq.util.json.dumps(data, indent = indent, sort_keys = sort_keys)

Render this document to JSON.

def to_tex(self, **kwargs: Any) -> str:
75    def to_tex(self, **kwargs: typing.Any) -> str:
76        """ Render this document to TeX. """
77
78        return self._render(quizcomp.model.constants.Format.TEX, **kwargs)

Render this document to TeX.

def to_text(self, **kwargs: Any) -> str:
80    def to_text(self, **kwargs: typing.Any) -> str:
81        """ Render this document to simple text. """
82
83        return self._render(quizcomp.model.constants.Format.TEXT, **kwargs)

Render this document to simple text.

def to_html(self, **kwargs: Any) -> str:
85    def to_html(self, **kwargs: typing.Any) -> str:
86        """ Render this document to HTML. """
87
88        return self._render(quizcomp.model.constants.Format.HTML, **kwargs)

Render this document to HTML.

def collect_placeholders(self) -> Set[str]:
111    def collect_placeholders(self) -> typing.Set[str]:
112        """
113        Fetch all the answer placeholders in this document.
114        """
115
116        tokens = self._collect_tokens(self._tokens, 'placeholder')
117        return {token.content for token in tokens}

Fetch all the answer placeholders in this document.

def collect_images(self) -> List[markdown_it.token.Token]:
119    def collect_images(self) -> typing.List[markdown_it.token.Token]:
120        """
121        Fetch all the image tokens in this document.
122        """
123
124        return self._collect_tokens(self._tokens, 'image')

Fetch all the image tokens in this document.

def is_empty(self) -> bool:
145    def is_empty(self) -> bool:
146        """ Check if this document contains any content (tokens). """
147
148        return (len(self._tokens) == 0)

Check if this document contains any content (tokens).

def to_format(self, format: quizcomp.model.constants.Format, **kwargs: Any) -> str:
155    def to_format(self, format: quizcomp.model.constants.Format, **kwargs: typing.Any) -> str:
156        """ Convert this document to the specified format. """
157
158        formatter = getattr(self, 'to_' + format.value)
159        if (formatter is None):
160            raise ValueError(f"Unknown format '{format.value}'.")
161
162        return str(formatter(**kwargs))

Convert this document to the specified format.

def get_ast(self) -> quizcomp.parser.ast.ASTNode:
164    def get_ast(self) -> quizcomp.parser.ast.ASTNode:
165        """
166        Get a representation of this document's AST.
167        """
168
169        return quizcomp.parser.ast.build(self._tokens)

Get a representation of this document's AST.

def to_pod( self, context: Optional[edq.util.common.SerializationContext] = None) -> str:
174    def to_pod(self,
175            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
176            ) -> str:
177        return self.text

Get a POD representation of this object.

The default implementation will convert to a dict (similar to a DictSerializer).

@classmethod
def parse_text( cls, text: Union[str, ParsedDocument], context: Optional[edq.util.common.SerializationContext] = None) -> ParsedDocument:
179    @classmethod
180    def parse_text(cls,
181            text: typing.Union[str, 'ParsedDocument'],
182            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
183            ) -> 'ParsedDocument':
184        """
185        Parse some text into a document.
186        If the text is already a document, that same document will be returned.
187        """
188
189        if (isinstance(text, ParsedDocument)):
190            return text
191
192        if (context is None):
193            context = edq.util.serial.SerializationContext()
194
195        text, tokens = quizcomp.parser.render._parse_text(text)
196        return quizcomp.parser.document.ParsedDocument(text, tokens, context)

Parse some text into a document. If the text is already a document, that same document will be returned.

@classmethod
def parse_file( cls, raw_path: str, context: Optional[edq.util.common.SerializationContext] = None) -> ParsedDocument:
198    @classmethod
199    def parse_file(cls,
200            raw_path: str,
201            context: typing.Union[edq.util.serial.SerializationContext, None] = None,
202            ) -> 'ParsedDocument':
203        """
204        Parse a text file into a document.
205
206        If a context is provided,
207        a copy will be made with the base dir and source path updated.
208        """
209
210        if (context is None):
211            context = edq.util.serial.SerializationContext()
212        else:
213            context = context.copy()
214
215        if (context.base_dir is None):
216            context.base_dir = os.path.dirname(os.path.abspath(raw_path))
217
218        # Prepend the base dir if the path is not absolute.
219        if (not os.path.isabs(raw_path)):
220            raw_path = os.path.join(context.base_dir, raw_path)
221
222        context.source_path = os.path.abspath(raw_path)
223        context.base_dir = os.path.dirname(context.source_path)
224
225        if (not os.path.isfile(context.source_path)):
226            raise ValueError(f"Path to parse ('{raw_path}') does not exist or is not a file.")
227
228        text = edq.util.dirent.read_file(context.source_path)
229
230        return cls.parse_text(text, context)

Parse a text file into a document.

If a context is provided, a copy will be made with the base dir and source path updated.