diff --git a/elixir/lib.py b/elixir/lib.py
index 8fcf2be3..82afc5fc 100755
--- a/elixir/lib.py
+++ b/elixir/lib.py
@@ -19,6 +19,7 @@
# along with Elixir. If not, see .
import sys
+import re
import logging
import subprocess, os
@@ -46,6 +47,23 @@ def scriptLines(*args, env=None):
del p[-1]
return p
+tokenize_regex_D = re.compile(rb'''((/\*.*?\*/|//.*?\001|[^']"(\\.|.)*?"|# *include *<.*?>|[^\w-])+)([\w-]+)?''')
+tokenize_regex = re.compile(rb'''((/\*.*?\*/|//.*?\001|[^']"(\\.|.)*?"|# *include *<.*?>|\W)+)(\w+)?''')
+
+def tokenizeBlob(hash, family):
+ p = subprocess.run(('git', 'cat-file', 'blob', hash), stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, cwd=os.environ['LXR_REPO_DIR'])
+ r = tokenize_regex_D if family == 'D' else tokenize_regex
+ for k in r.findall(p.stdout.replace(b'\n', b'\001')):
+ yield k[0]
+ yield k[3]
+
+def tokenizeFile(ver, file, family, env=None):
+ data = script('get-file', ver, file, env=env).replace(b'\n', b'\001')
+ r = tokenize_regex_D if family == 'D' else tokenize_regex
+ for k in r.findall(data):
+ yield k[0]
+ yield k[3]
+
def unescape(bstr):
subs = (
('\1','\n'),
diff --git a/elixir/query.py b/elixir/query.py
index e3e4ffd3..2d4ea1c1 100755
--- a/elixir/query.py
+++ b/elixir/query.py
@@ -18,7 +18,7 @@
# You should have received a copy of the GNU Affero General Public License
# along with Elixir. If not, see .
-from .lib import script, scriptLines, decode
+from .lib import script, scriptLines, decode, tokenizeFile
from . import lib
from . import data
import os
@@ -113,7 +113,7 @@ def get_tokenized_file(self, version, path):
assert family in lib.CACHED_DEFINITIONS_FAMILIES, f"family {family} must have its definitions cached"
buffer = BytesIO()
- tokens = self.scriptLines('tokenize-file', version, path, family)
+ tokens = tokenizeFile(version, path, family, env=self.getEnv())
even = True
prefix = b''
@@ -334,4 +334,3 @@ def get_idents_defs(self, version, ident, family):
symbol_doccomments.append(SymbolInstance(path, docline))
return symbol_definitions, symbol_references, symbol_doccomments, True
-
diff --git a/script.sh b/script.sh
index 3bbff2a7..3cbb384b 100755
--- a/script.sh
+++ b/script.sh
@@ -94,27 +94,6 @@ get_dir()
sort -t ' ' -k 1,1r -k 2,2
}
-tokenize_file()
-{
- if [ "$opt1" = -b ]; then
- ref=$opt2
- else
- v=`echo $opt1 | version_rev`
- ref="$v:`denormalize $opt2`"
- fi
-
- if [ $opt3 = "D" ]; then #Don't cut around '-' in devicetrees
- regex='s%((/\*.*?\*/|//.*?\001|[^'"'"']"(\\.|.)*?"|# *include *<.*?>|[^\w-])+)([\w-]+)?%\1\n\4\n%g'
- else
- regex='s%((/\*.*?\*/|//.*?\001|[^'"'"']"(\\.|.)*?"|# *include *<.*?>|\W)+)(\w+)?%\1\n\4\n%g'
- fi
-
- git cat-file blob $ref 2>/dev/null |
- tr '\n' '\1' |
- perl -pe "$regex" |
- head -n -1
-}
-
list_blobs()
{
v=`echo $opt2 | version_rev`
@@ -271,10 +250,6 @@ case $cmd in
list_blobs
;;
- tokenize-file)
- tokenize_file
- ;;
-
untokenize)
untokenize
;;
diff --git a/update.py b/update.py
index 9d84ff31..fa622ccf 100755
--- a/update.py
+++ b/update.py
@@ -26,7 +26,7 @@
from threading import Thread, Lock, Event, Condition
import elixir.lib as lib
-from elixir.lib import script, scriptLines
+from elixir.lib import script, scriptLines, tokenizeBlob
import elixir.data as data
from elixir.data import PathList
from find_compatible_dts import FindCompatibleDTS
@@ -316,7 +316,7 @@ def update_references(self, idxes):
if family == 'K':
prefix = b'CONFIG_'
- tokens = scriptLines('tokenize-file', '-b', hash, family)
+ tokens = tokenizeBlob(hash, family)
even = True
line_num = 1
idents = {}