uglify/llex.lua

1
--[[--------------------------------------------------------------------
2
 
3
  llex.lua: Lua 5.1 lexical analyzer in Lua
4
  This file is part of LuaSrcDiet, based on Yueliang material.
5
 
6
  Copyright (c) 2008 Kein-Hong Man <khman@users.sf.net>
7
  The COPYRIGHT file describes the conditions
8
  under which this software may be distributed.
9
 
10
  See the ChangeLog for more information.
11
 
12
----------------------------------------------------------------------]]
13
 
14
--[[--------------------------------------------------------------------
15
-- NOTES:
16
-- * This is a version of the native 5.1.x lexer from Yueliang 0.4.0,
17
--   with significant modifications to handle LuaSrcDiet's needs:
18
--   (1) llex.error is an optional error function handler
19
--   (2) seminfo for strings include their delimiters and no
20
--       translation operations are performed on them
21
-- * ADDED shbang handling has been added to support executable scripts
22
-- * NO localized decimal point replacement magic
23
-- * NO limit to number of lines
24
-- * NO support for compatible long strings (LUA_COMPAT_LSTR)
25
-- * Please read technotes.txt for more technical details.
26
----------------------------------------------------------------------]]
27
 
28
local base = _G
29
local string = require "string"
30
--module "llex"
31
 
32
local find = string.find
33
local match = string.match
34
local sub = string.sub
35
 
36
----------------------------------------------------------------------
37
-- initialize keyword list, variables
38
----------------------------------------------------------------------
39
 
40
local kw = {}
41
for v in string.gmatch([[
42
and break do else elseif end false for function if in
43
local nil not or repeat return then true until while]], "%S+") do
44
  kw[v] = true
45
end
46
 
47
-- NOTE: see init() for module variables (externally visible):
48
--       tok, seminfo, tokln
49
 
50
 local z = '',                        -- source
51
  sourceid = '',          -- name of source
52
  I = 1,                         -- lexer's position in source
53
  buff = '',
54
  ln = 1,                        -- line number
55
  tok = {},                      -- lexed token list*
56
  seminfo = {},                  -- lexed semantic information list*
57
  tokln = {},                    -- line numbers for messages*
58
 
59
----------------------------------------------------------------------
60
-- add information to token listing
61
----------------------------------------------------------------------
62
 
63
local function addtoken(token, info)
64
  local i = #tok + 1
65
  tok[i] = token
66
  seminfo[i] = info
67
  tokln[i] = ln
68
end
69
 
70
----------------------------------------------------------------------
71
-- handles line number incrementation and end-of-line characters
72
----------------------------------------------------------------------
73
 
74
local function inclinenumber(i, is_tok)
75
  local sub = sub
76
  local old = sub(z, i, i)
77
  i = i + 1  -- skip '\n' or '\r'
78
  local c = sub(z, i, i)
79
  if (c == "\n" or c == "\r") and (c ~= old) then
80
    i = i + 1  -- skip '\n\r' or '\r\n'
81
    old = old..c
82
  end
83
  if is_tok then addtoken("TK_EOL", old) end
84
  ln = ln + 1
85
  I = i
86
  return i
87
end
88
 
89
----------------------------------------------------------------------
90
-- initialize lexer for given source _z and source name _sourceid
91
----------------------------------------------------------------------
92
 
93
function init(_z, _sourceid)
94
  z = _z                        -- source
95
  sourceid = _sourceid          -- name of source
96
  I = 1                         -- lexer's position in source
97
  ln = 1                        -- line number
98
  tok = {}                      -- lexed token list*
99
  seminfo = {}                  -- lexed semantic information list*
100
  tokln = {}                    -- line numbers for messages*
101
                                -- (*) externally visible thru' module
102
  --------------------------------------------------------------------
103
  -- initial processing (shbang handling)
104
  --------------------------------------------------------------------
105
  local p, _, q, r = find(z, "^(#[^\r\n]*)(\r?\n?)")
106
  if p then                             -- skip first line
107
    I = I + #q
108
    addtoken("TK_COMMENT", q)
109
    if #r > 0 then inclinenumber(I, true) end
110
  end
111
end
112
 
113
----------------------------------------------------------------------
114
-- returns a chunk name or id, no truncation for long names
115
----------------------------------------------------------------------
116
 
117
function chunkid()
118
  if sourceid and match(sourceid, "^[=@]") then
119
    return sub(sourceid, 2)  -- remove first char
120
  end
121
  return "[string]"
122
end
123
 
124
----------------------------------------------------------------------
125
-- formats error message and throws error
126
-- * a simplified version, does not report what token was responsible
127
----------------------------------------------------------------------
128
 
129
function errorline(s, line)
130
  local e = error or base.error
131
  e(string.format("%s:%d: %s", chunkid(), line or ln, s))
132
end
133
local errorline = errorline
134
 
135
------------------------------------------------------------------------
136
-- count separators ("=") in a long string delimiter
137
------------------------------------------------------------------------
138
 
139
local function skip_sep(i)
140
  local sub = sub
141
  local s = sub(z, i, i)
142
  i = i + 1
143
  local count = #match(z, "=*", i)  -- note, take the length
144
  i = i + count
145
  I = i
146
  return (sub(z, i, i) == s) and count or (-count) - 1
147
end
148
 
149
----------------------------------------------------------------------
150
-- reads a long string or long comment
151
----------------------------------------------------------------------
152
 
153
local function read_long_string(is_str, sep)
154
  local i = I + 1  -- skip 2nd '['
155
  local sub = sub
156
  local c = sub(z, i, i)
157
  if c == "\r" or c == "\n" then  -- string starts with a newline?
158
    i = inclinenumber(i)  -- skip it
159
  end
160
  local j = i
161
  while true do
162
    local p, q, r = find(z, "([\r\n%]])", i) -- (long range)
163
    if not p then
164
      errorline(is_str and "unfinished long string" or
165
                "unfinished long comment")
166
    end
167
    i = p
168
    if r == "]" then                    -- delimiter test
169
      if skip_sep(i) == sep then
170
        buff = sub(z, buff, I)
171
        I = I + 1  -- skip 2nd ']'
172
        return buff
173
      end
174
      i = I
175
    else                                -- newline
176
      buff = buff.."\n"
177
      i = inclinenumber(i)
178
    end
179
  end--while
180
end
181
 
182
----------------------------------------------------------------------
183
-- reads a string
184
----------------------------------------------------------------------
185
 
186
local function read_string(del)
187
  local i = I
188
  local find = find
189
  local sub = sub
190
  while true do
191
    local p, q, r = find(z, "([\n\r\\\"\'])", i) -- (long range)
192
    if p then
193
      if r == "\n" or r == "\r" then
194
        errorline("unfinished string")
195
      end
196
      i = p
197
      if r == "\\" then                         -- handle escapes
198
        i = i + 1
199
        r = sub(z, i, i)
200
        if r == "" then break end -- (EOZ error)
201
        p = find("abfnrtv\n\r", r, 1, true)
202
        ------------------------------------------------------
203
        if p then                               -- special escapes
204
          if p > 7 then
205
            i = inclinenumber(i)
206
          else
207
            i = i + 1
208
          end
209
        ------------------------------------------------------
210
        elseif find(r, "%D") then               -- other non-digits
211
          i = i + 1
212
        ------------------------------------------------------
213
        else                                    -- \xxx sequence
214
          local p, q, s = find(z, "^(%d%d?%d?)", i)
215
          i = q + 1
216
          if s + 1 > 256 then -- UCHAR_MAX
217
            errorline("escape sequence too large")
218
          end
219
        ------------------------------------------------------
220
        end--if p
221
      else
222
        i = i + 1
223
        if r == del then                        -- ending delimiter
224
          I = i
225
          return sub(z, buff, i - 1)            -- return string
226
        end
227
      end--if r
228
    else
229
      break -- (error)
230
    end--if p
231
  end--while
232
  errorline("unfinished string")
233
end
234
 
235
------------------------------------------------------------------------
236
-- main lexer function
237
------------------------------------------------------------------------
238
 
239
function llex()
240
  local find = find
241
  local match = match
242
  while true do--outer
243
    local i = I
244
    -- inner loop allows break to be used to nicely section tests
245
    while true do--inner
246
      ----------------------------------------------------------------
247
      local p, _, r = find(z, "^([_%a][_%w]*)", i)
248
      if p then
249
        I = i + #r
250
        if kw[r] then
251
          addtoken("TK_KEYWORD", r)             -- reserved word (keyword)
252
        else
253
          addtoken("TK_NAME", r)                -- identifier
254
        end
255
        break -- (continue)
256
      end
257
      ----------------------------------------------------------------
258
      local p, _, r = find(z, "^(%.?)%d", i)
259
      if p then                                 -- numeral
260
        if r == "." then i = i + 1 end
261
        local _, q, r = find(z, "^%d*[%.%d]*([eE]?)", i)
262
        i = q + 1
263
        if #r == 1 then                         -- optional exponent
264
          if match(z, "^[%+%-]", i) then        -- optional sign
265
            i = i + 1
266
          end
267
        end
268
        local _, q = find(z, "^[_%w]*", i)
269
        I = q + 1
270
        local v = sub(z, p, q)                  -- string equivalent
271
        if not base.tonumber(v) then            -- handles hex test also
272
          errorline("malformed number")
273
        end
274
        addtoken("TK_NUMBER", v)
275
        break -- (continue)
276
      end
277
      ----------------------------------------------------------------
278
      local p, q, r, t = find(z, "^((%s)[ \t\v\f]*)", i)
279
      if p then
280
        if t == "\n" or t == "\r" then          -- newline
281
          inclinenumber(i, true)
282
        else
283
          I = q + 1                             -- whitespace
284
          addtoken("TK_SPACE", r)
285
        end
286
        break -- (continue)
287
      end
288
      ----------------------------------------------------------------
289
      local r = match(z, "^%p", i)
290
      if r then
291
        buff = i
292
        local p = find("-[\"\'.=<>~", r, 1, true)
293
        if p then
294
          -- two-level if block for punctuation/symbols
295
          --------------------------------------------------------
296
          if p <= 2 then
297
            if p == 1 then                      -- minus
298
              local c = match(z, "^%-%-(%[?)", i)
299
              if c then
300
                i = i + 2
301
                local sep = -1
302
                if c == "[" then
303
                  sep = skip_sep(i)
304
                end
305
                if sep >= 0 then                -- long comment
306
                  addtoken("TK_LCOMMENT", read_long_string(false, sep))
307
                else                            -- short comment
308
                  I = find(z, "[\n\r]", i) or (#z + 1)
309
                  addtoken("TK_COMMENT", sub(z, buff, I - 1))
310
                end
311
                break -- (continue)
312
              end
313
              -- (fall through for "-")
314
            else                                -- [ or long string
315
              local sep = skip_sep(i)
316
              if sep >= 0 then
317
                addtoken("TK_LSTRING", read_long_string(true, sep))
318
              elseif sep == -1 then
319
                addtoken("TK_OP", "[")
320
              else
321
                errorline("invalid long string delimiter")
322
              end
323
              break -- (continue)
324
            end
325
          --------------------------------------------------------
326
          elseif p <= 5 then
327
            if p < 5 then                       -- strings
328
              I = i + 1
329
              addtoken("TK_STRING", read_string(r))
330
              break -- (continue)
331
            end
332
            r = match(z, "^%.%.?%.?", i)        -- .|..|... dots
333
            -- (fall through)
334
          --------------------------------------------------------
335
          else                                  -- relational
336
            r = match(z, "^%p=?", i)
337
            -- (fall through)
338
          end
339
        end
340
        I = i + #r
341
        addtoken("TK_OP", r)  -- for other symbols, fall through
342
        break -- (continue)
343
      end
344
      ----------------------------------------------------------------
345
      local r = sub(z, i, i)
346
      if r ~= "" then
347
        I = i + 1
348
        addtoken("TK_OP", r)                    -- other single-char tokens
349
        break
350
      end
351
      addtoken("TK_EOS", "")                    -- end of stream,
352
      return                                    -- exit here
353
      ----------------------------------------------------------------
354
    end--while inner
355
  end--while outer
356
end
357
 
358
return {
359
init = init,
360
llex = llex
361
}