aboutsummaryrefslogtreecommitdiff
path: root/source/lexer.elna
blob: 318d7401c9f41ccd6d1e8fc63db6a582ecf5514a (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
(* This Source Code Form is subject to the terms of the Mozilla Public License,
  v. 2.0. If a copy of the MPL was not distributed with this file, You can
  obtain one at https://mozilla.org/MPL/2.0/. *)

import cstdio, common

const
  CHUNK_SIZE := 85536u

type
	ElnaLexerState = (
		start,
		colon,
		identifier,
		decimal,
		leading_zero,
		greater,
		minus,
		left_paren,
		less,
		dot,
		comment,
		closing_comment,
		character,
		character_escape,
		string,
		string_escape,
		number_sign,
		trait,
		finish
	)
  ElnaLexerToken* = record
    kind: ElnaLexerKind;
		start: String; (* DEPRECATED *)
		position: ElnaPosition
  end
  ElnaLexerBooleanToken* = record(ElnaLexerToken)
    value: Bool
  end
  ElnaLexerCharacterToken* = record(ElnaLexerToken)
    value: Char
  end
  ElnaLexerStringToken* = record(ElnaLexerToken)
    value: String
  end
  ElnaLexerIntegerToken* = record(ElnaLexerToken)
    value: Int
  end

  TransitionAction = proc(lexer: ^Lexer; token: ^ElnaLexerToken)
  (* DEPRECATED should be replaced with TransitionAction (but keeping the type name). *)
	ElnaLexerAction = (none, accumulate, skip, single, eof, finalize, composite, key_id, integer, delimited)

  ElnaLexerTransition = record
    action: ElnaLexerAction;
    next_state: ElnaLexerState
  end

	ElnaLexerCursor = record
		state: ElnaLexerState;
		start: ^Char;
		finish: ^Char;
		token: ^ElnaLexerToken;
		position: ElnaPosition
	end

  BufferPosition* = record
    iterator: ^Char;
    location: ElnaLocation
  end
  Lexer* = record
    input: ^FILE;
    buffer: ^Char;
    size: Word;
    length: Word;
    start: BufferPosition;
    current: BufferPosition
  end
	ElnaLexerKind* = (
		identifier,
		_const,
		_var,
		_proc,
		_type,
		_begin,
		_end,
		_if,
		_then,
		_else,
		_elsif,
		_extern,
		_record,
		boolean,
		null,
		and,
		_or,
		_xor,
		not,
		_return,
		_cast,
		trait,
		left_paren,
		right_paren,
		left_square,
		right_square,
		greater_equal,
		less_equal,
		greater_than,
		less_than,
		not_equal,
		equals,
		semicolon,
		dot,
		comma,
		plus,
		_import,
		minus,
		multiplication,
		division,
		remainder,
		assignment,
		colon,
		hat,
		at,
		comment,
		string,
		character,
		integer,
		word,
		_while,
    _defer,
    exclamation,
    shift_right,
    shift_left,
    pipe,
    _case,
    _do,
    _of,
		eof
	)
	(**
	 * Classification table assigns each possible character to a group (class). All
	 * characters of the same group a handled equivalently.
	 *)
	ElnaLexerClass = (
		invalid,
		digit,
		alpha,
		space,
		colon,
		equals,
		left_paren,
		right_paren,
		asterisk,
		backslash,
		single,
		hex,
		zero,
		x,
		eof,
		dot,
		minus,
		single_quote,
		double_quote,
		greater,
		less,
		other,
		number_sign
	)

end.