feat(lint): libclang-backed token layer

Adds LintContext::Tokens() and friends, backed by clang_tokenize, as the
substrate the rules will move onto. Nothing consumes it yet.

libclang is dlopen'd rather than linked: -lclang would break the mingw and
MSVC cross-builds at link time and would put a libclang.so.NN runtime
dependency into the otherwise self-contained release tarballs. The clang-c
header is used for its declarations only, and the function-pointer table is
typed with decltype so the signatures cannot drift from the real API.

Three properties this buys that the hand-rolled scanners could not have:

  - a raw string literal or block comment is ONE token, so the documented
    "raw string literals are not recognized" limitation goes away;
  - `//` inside a literal is not a comment, so LineHasComment() replaces the
    Line(n).contains("//") probes that false-positive on it;
  - tokens cover preprocessor branches that are inactive for the host, since
    clang_tokenize lexes rather than evaluates #if. Token rules therefore
    keep seeing every platform's code, which an AST could not offer.

The parse backing the tokenizer is expected to fail on module units — no
PCMs, no build flags — and that is fine, because lexing has no semantic
prerequisites. Verified in the new tests.

LintSummary::Clean() now counts `errors`. It previously ignored them, so an
infrastructure failure that produced no findings reported clean and exited
0; a missing libclang would have been exactly that.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Jorijn van der Graaf 2026-07-27 02:54:38 +02:00
commit 5888af29ed
4 changed files with 387 additions and 2 deletions

View file

@ -2,6 +2,12 @@
// SPDX-FileCopyrightText: Copyright (C) 2026 Catcrafts®
module;
#include <clang-c/Index.h>
#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32)
#include <windows.h>
#else
#include <dlfcn.h>
#endif
export module Crafter.Build:Lint_impl;
import std;
import :Lint;
@ -12,6 +18,203 @@ namespace fs = std::filesystem;
using namespace Crafter;
namespace {
// ---------------- libclang ----------------
//
// libclang is loaded at runtime rather than linked. Linking -lclang would
// break the mingw and MSVC cross-builds at link time and would put a
// libclang.so.NN runtime dependency into the otherwise self-contained
// release tarballs; the clang-c header is used for its declarations only,
// and every call goes through a pointer resolved here. Failure to load is
// a hard error surfaced once by RunLint — there is deliberately no second,
// weaker lexer to fall back to, because two engines disagreeing about what
// is a comment is a worse failure than not running.
#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32)
using LibHandle = HMODULE;
LibHandle OpenLibrary(const std::string& name) { return LoadLibraryA(name.c_str()); }
void* LibrarySymbol(LibHandle handle, const std::string& name) { return reinterpret_cast<void*>(GetProcAddress(handle, name.c_str())); }
constexpr std::string_view LibClangNames[] = {"libclang.dll", "clang.dll"};
#else
using LibHandle = void*;
LibHandle OpenLibrary(const std::string& name) { return dlopen(name.c_str(), RTLD_NOW | RTLD_LOCAL); }
void* LibrarySymbol(LibHandle handle, const std::string& name) { return dlsym(handle, name.c_str()); }
constexpr std::string_view LibClangNames[] = {
"libclang.so", "libclang.so.22.1", "libclang.so.21.1", "libclang.so.20.1",
"libclang.so.1", "libclang.dylib",
};
#endif
// Signatures come from decltype on the header's declarations, so they can
// never drift from the real API. decltype is unevaluated, so naming the
// functions here does not create a link-time reference to them.
struct LibClang {
LibHandle handle = nullptr;
std::string error; // non-empty exactly when handle is null
decltype(&clang_createIndex) CreateIndex = nullptr;
decltype(&clang_disposeIndex) DisposeIndex = nullptr;
decltype(&clang_parseTranslationUnit) ParseTranslationUnit = nullptr;
decltype(&clang_disposeTranslationUnit) DisposeTranslationUnit = nullptr;
decltype(&clang_getFile) GetFile = nullptr;
decltype(&clang_getLocationForOffset) GetLocationForOffset = nullptr;
decltype(&clang_getRange) GetRange = nullptr;
decltype(&clang_getRangeStart) GetRangeStart = nullptr;
decltype(&clang_getRangeEnd) GetRangeEnd = nullptr;
decltype(&clang_getFileLocation) GetFileLocation = nullptr;
decltype(&clang_tokenize) Tokenize = nullptr;
decltype(&clang_disposeTokens) DisposeTokens = nullptr;
decltype(&clang_getTokenKind) GetTokenKind = nullptr;
decltype(&clang_getTokenExtent) GetTokenExtent = nullptr;
};
LibClang LoadLibClang() {
LibClang lib;
std::vector<std::string> tried;
// CRAFTER_BUILD_LIBCLANG pins an exact path, mirroring the LIBCXX_DIR /
// CRAFTER_MINGW_DIR overrides used elsewhere. It is exclusive: pointing
// it at a broken path must fail loudly rather than quietly succeed with
// some other libclang, or the override is useless for diagnosing which
// library is actually in play.
std::vector<std::string> candidates;
if (const char* pinned = std::getenv("CRAFTER_BUILD_LIBCLANG"); pinned && *pinned) {
candidates.emplace_back(pinned);
} else {
for (std::string_view name : LibClangNames) candidates.emplace_back(name);
}
auto join = [](const std::vector<std::string>& parts) {
std::string joined;
for (const std::string& part : parts) {
if (!joined.empty()) joined += ", ";
joined += part;
}
return joined;
};
for (const std::string& name : candidates) {
lib.handle = OpenLibrary(name);
if (lib.handle) break;
tried.push_back(name);
}
if (!lib.handle) {
lib.error = std::format("could not load libclang (tried {}); install clang, or point CRAFTER_BUILD_LIBCLANG at it", join(tried));
return lib;
}
std::vector<std::string> missing;
auto bind = [&](auto& slot, const std::string& name) {
slot = reinterpret_cast<std::remove_reference_t<decltype(slot)>>(LibrarySymbol(lib.handle, name));
if (!slot) missing.push_back(name);
};
bind(lib.CreateIndex, "clang_createIndex");
bind(lib.DisposeIndex, "clang_disposeIndex");
bind(lib.ParseTranslationUnit, "clang_parseTranslationUnit");
bind(lib.DisposeTranslationUnit, "clang_disposeTranslationUnit");
bind(lib.GetFile, "clang_getFile");
bind(lib.GetLocationForOffset, "clang_getLocationForOffset");
bind(lib.GetRange, "clang_getRange");
bind(lib.GetRangeStart, "clang_getRangeStart");
bind(lib.GetRangeEnd, "clang_getRangeEnd");
bind(lib.GetFileLocation, "clang_getFileLocation");
bind(lib.Tokenize, "clang_tokenize");
bind(lib.DisposeTokens, "clang_disposeTokens");
bind(lib.GetTokenKind, "clang_getTokenKind");
bind(lib.GetTokenExtent, "clang_getTokenExtent");
if (!missing.empty()) {
lib.handle = nullptr;
lib.error = std::format("loaded {} but it is missing {}", candidates.front(), join(missing));
}
return lib;
}
const LibClang& Clang() {
static const LibClang Lib = LoadLibClang();
return Lib;
}
// The -x language for a source file, or empty when we must not lex it.
// .cppm needs c++-module explicitly: libclang does not infer a module unit
// from the extension and silently treats every flag as a linker input if
// left to guess. Shaders and data files return empty — lexing GLSL as C++
// yields plausible-looking nonsense.
std::string_view LexLanguage(const fs::path& file) {
std::string ext = file.extension().string();
if (ext == ".cppm" || ext == ".ixx") return "c++-module";
if (ext == ".cpp" || ext == ".cc" || ext == ".cxx" || ext == ".h" || ext == ".hpp" || ext == ".cu") return "c++";
if (ext == ".c") return "c";
return {};
}
LintTokenKind MapTokenKind(CXTokenKind kind) {
switch (kind) {
case CXToken_Punctuation: return LintTokenKind::Punctuation;
case CXToken_Keyword: return LintTokenKind::Keyword;
case CXToken_Identifier: return LintTokenKind::Identifier;
case CXToken_Literal: return LintTokenKind::Literal;
case CXToken_Comment: return LintTokenKind::Comment;
}
return LintTokenKind::Punctuation;
}
// Lex `content` as if it were `file`, returning tokens in source order.
//
// The buffer is handed over as an unsaved file, so a transform's in-memory
// edits are what get lexed — never the stale bytes on disk. The parse is
// expected to fail (a module unit's `import std;` cannot resolve without
// PCMs, and we deliberately do not supply the build's flags here); that
// does not matter, because clang_tokenize re-lexes the buffer and lexing
// has no semantic prerequisites. SingleFileParse keeps it from chasing
// #includes it does not need.
std::vector<LintToken> LexFile(const fs::path& file, const std::string& content) {
std::string_view language = LexLanguage(file);
if (language.empty()) return {};
const LibClang& lc = Clang();
if (!lc.handle) return {};
std::string path = file.string();
std::string languageArg = std::format("-x{}", language);
std::string standardArg = language == "c" ? "-std=c23" : "-std=c++26";
// clang's argv is char* by contract; keep the raw pointers confined to
// this call rather than letting them into any signature of ours.
std::array<const char*, 4> args{languageArg.c_str(), standardArg.c_str(), "-ferror-limit=0", "-w"};
CXUnsavedFile unsaved{};
unsaved.Filename = path.c_str();
unsaved.Contents = content.data();
unsaved.Length = static_cast<std::uint32_t>(content.size());
CXIndex index = lc.CreateIndex(0, 0);
if (!index) return {};
CXTranslationUnit tu = lc.ParseTranslationUnit(index, path.c_str(), args.data(), static_cast<std::int32_t>(args.size()), &unsaved, 1, CXTranslationUnit_SingleFileParse | CXTranslationUnit_SkipFunctionBodies | CXTranslationUnit_KeepGoing);
if (!tu) {
lc.DisposeIndex(index);
return {};
}
std::vector<LintToken> tokens;
if (CXFile cxFile = lc.GetFile(tu, path.c_str())) {
CXSourceRange whole = lc.GetRange(lc.GetLocationForOffset(tu, cxFile, 0), lc.GetLocationForOffset(tu, cxFile, static_cast<std::uint32_t>(content.size())));
CXToken* raw = nullptr;
std::uint32_t count = 0;
lc.Tokenize(tu, whole, &raw, &count);
tokens.reserve(count);
for (std::uint32_t i = 0; i < count; ++i) {
CXSourceRange extent = lc.GetTokenExtent(tu, raw[i]);
std::uint32_t line = 0;
std::uint32_t column = 0;
std::uint32_t begin = 0;
std::uint32_t end = 0;
lc.GetFileLocation(lc.GetRangeStart(extent), nullptr, &line, &column, &begin);
lc.GetFileLocation(lc.GetRangeEnd(extent), nullptr, nullptr, nullptr, &end);
if (end < begin || begin > content.size()) continue;
tokens.push_back({MapTokenKind(lc.GetTokenKind(raw[i])), begin, std::min<std::size_t>(end - begin, content.size() - begin), line, column});
}
if (raw) lc.DisposeTokens(tu, raw, count);
}
lc.DisposeTranslationUnit(tu);
lc.DisposeIndex(index);
return tokens;
}
// Blank //-comments, /*...*/ comments and string/char literal bodies to
// spaces while copying '\n' through, so byte offsets and line numbers in
// the result match the original text. Raw string literals are not
@ -164,6 +367,30 @@ const std::string& LintContext::CommentStripped() {
return *commentStrippedCache;
}
std::span<const LintToken> LintContext::Tokens() {
if (!tokenCache) tokenCache = LexFile(file, content);
return *tokenCache;
}
std::string_view LintContext::TokenText(const LintToken& token) const {
if (token.offset >= content.size()) return {};
return std::string_view(content).substr(token.offset, token.length);
}
std::span<const LintToken> LintContext::TokensOnLine(std::size_t line) {
// Tokens come back in source order, so one line's tokens are a contiguous
// run and can be bracketed by binary search.
std::span<const LintToken> all = Tokens();
auto begin = std::ranges::lower_bound(all, line, {}, &LintToken::line);
auto end = std::ranges::upper_bound(all, line, {}, &LintToken::line);
return all.subspan(static_cast<std::size_t>(begin - all.begin()), static_cast<std::size_t>(end - begin));
}
bool LintContext::LineHasComment(std::size_t line) {
std::span<const LintToken> onLine = TokensOnLine(line);
return std::ranges::any_of(onLine, [](const LintToken& t) { return t.kind == LintTokenKind::Comment; });
}
void LintContext::Report(std::size_t line, std::string message) {
sink->push_back({file, line, activeRule, std::move(message)});
}
@ -226,6 +453,7 @@ void LintContext::SetContent(std::string newContent) {
lines = SplitLines(content);
commentStrippedCache.reset();
suppressionsCache.reset(); // line numbers may have shifted — re-parse
tokenCache.reset(); // offsets refer to the old buffer — re-lex
}
void Configuration::AddLintRule(std::string name, std::function<void(LintContext&)> check) {
@ -235,6 +463,15 @@ void Configuration::AddLintRule(std::string name, std::function<void(LintContext
LintSummary Crafter::RunLint(Configuration& projectCfg, const RunLintOptions& opts) {
LintSummary summary;
// libclang backs the lexer every rule reads through, so a failed load is
// fatal rather than a downgrade: running the rules without it would report
// against a substrate that disagrees with the one they were written for.
if (const LibClang& lc = Clang(); !lc.handle) {
std::println(std::cerr, "lint: {}", lc.error);
++summary.errors;
return summary;
}
fs::path projectRoot = opts.projectFile.empty()
? fs::absolute(projectCfg.path)
: opts.projectFile.parent_path();

View file

@ -124,6 +124,31 @@ export namespace Crafter {
std::unordered_map<std::size_t, std::unordered_set<std::string>> lineRules;
};
// Lexical class of a LintToken, mirroring clang's token kinds one-to-one.
enum class LintTokenKind {
Punctuation,
Keyword,
Identifier,
Literal, // string, raw string, character, integer, floating literal
Comment, // // ... or /* ... */
};
// One token from LintContext::Tokens(). `offset`/`length` are byte offsets
// into LintContext::content and stay valid until the next SetContent.
//
// A raw string literal or a block comment is ONE token and may span lines,
// which is exactly what the hand-rolled scanners could not represent. The
// stream also covers text inside preprocessor branches that are inactive
// for the host — clang_tokenize lexes, it does not evaluate #if — so token
// rules see every platform's code, not just the one being built.
struct LintToken {
LintTokenKind kind = LintTokenKind::Punctuation;
std::size_t offset = 0;
std::size_t length = 0;
std::size_t line = 0; // 1-based, of the token's first byte
std::size_t column = 0; // 1-based, of the token's first byte
};
// Per-file view handed to each LintRule's check callback. Every member
// function is out-of-line and CRAFTER_API (defined in Crafter.Build:Lint's
// implementation unit) because rule lambdas execute from the user's
@ -165,11 +190,31 @@ export namespace Crafter {
// re-parsed after SetContent.
CRAFTER_API bool Suppressed(std::string_view rule, std::size_t line);
// The file lexed by clang, in source order. Built on first call, cached
// per file, re-lexed after SetContent. Empty for extensions that are
// not C or C++ (shaders): lexing GLSL as C++ would produce nonsense.
//
// Prefer this to scanning characters. It is the only view that gets
// raw strings, line splices, digraphs and nested quoting right, and
// offsets index straight into `content`, so a transform can find in
// the token stream and edit in place.
CRAFTER_API std::span<const LintToken> Tokens();
// The token's own bytes: content.substr(tok.offset, tok.length).
CRAFTER_API std::string_view TokenText(const LintToken& token) const;
// Tokens whose FIRST byte is on `line` (1-based). A token that starts
// earlier and spans into `line` — a raw string, a block comment — is
// not included; ask Tokens() directly when that matters.
CRAFTER_API std::span<const LintToken> TokensOnLine(std::size_t line);
// True when a comment token starts on `line`. Replaces `Line(n).contains("//")`,
// which false-positives on a `//` inside a string literal.
CRAFTER_API bool LineHasComment(std::size_t line);
// Driver wiring — set by RunLint before each check call. Not for rules.
std::string activeRule;
std::vector<LintFinding>* sink = nullptr;
std::optional<std::string> commentStrippedCache;
std::optional<LintSuppressions> suppressionsCache;
std::optional<std::vector<LintToken>> tokenCache;
};
// A named lint rule: `check` runs once per (rule, file) over the project's

View file

@ -39,10 +39,13 @@ export namespace Crafter {
std::vector<std::filesystem::path> changedFiles;
std::size_t filesLinted = 0;
std::size_t rulesRun = 0; // rules remaining after glob filter
std::size_t errors = 0; // rule exceptions + write failures
// Rule exceptions, write failures, and infrastructure failures such as
// libclang not loading. Counted in Clean() so a run that could not do
// its job never looks like a run that found nothing.
std::size_t errors = 0;
bool noRulesDefined = false; // project registered no rules at all
// Host-side only (like TestSummary::AllPassed), safe as in-class inline.
bool Clean() const { return findings.empty() && !noRulesDefined; }
bool Clean() const { return findings.empty() && !noRulesDefined && errors == 0; }
};
// Run the project's lint rules over its own sources: module interfaces

View file

@ -427,6 +427,106 @@ int main() {
Check(s.Read("a") == "// lint-disable-file flag\nbad\n", "other rules still format");
}
// ---------------- token layer ----------------
//
// The source is spelled with escaped literals rather than a raw string so
// that this file stays lintable by the very rules under test; the scratch
// file it writes does contain a genuine multi-line raw string.
{
constexpr std::string_view Source =
"#ifdef CRAFTER_LINT_NEVER_DEFINED\n" // 1
"void HiddenBranch();\n" // 2
"#endif\n" // 3
"// a real comment\n" // 4
"int url = 1; // https://example.com\n" // 5
"auto raw = R\"raw(spans lines\n" // 6
" // not a comment\n" // 7
" int notADecl;\n" // 8
")raw\";\n"; // 9
Scratch s("tokens");
s.Write("f", Source);
Configuration cfg = s.Config({"f"});
cfg.AddLintRule("tokens", [](LintContext& ctx) {
std::span<const LintToken> toks = ctx.Tokens();
Check(!toks.empty(), "tokens: file lexes to a non-empty stream");
// Every token's offset/length must address its own bytes, or a
// transform editing at an offset would corrupt the file.
bool offsetsSound = true;
for (const LintToken& t : toks) {
if (t.offset + t.length > ctx.content.size() || ctx.TokenText(t).empty()) offsetsSound = false;
}
Check(offsetsSound, "tokens: every offset/length addresses real bytes");
// Ordered by offset, so binary search in TokensOnLine is valid.
bool ordered = std::ranges::is_sorted(toks, {}, &LintToken::offset);
Check(ordered, "tokens: stream is in source order");
// Inactive #ifdef branch is still lexed — this is what keeps token
// rules covering every platform, unlike an AST.
bool sawHidden = std::ranges::any_of(toks, [&](const LintToken& t) {
return t.kind == LintTokenKind::Identifier && ctx.TokenText(t) == "HiddenBranch";
});
Check(sawHidden, "tokens: inactive #ifdef branch is lexed");
// A multi-line raw string is exactly one Literal, comment markers
// and declarations inside it included.
auto isRaw = [&](const LintToken& t) { return ctx.TokenText(t).starts_with("R\"raw("); };
Check(std::ranges::count_if(toks, isRaw) == 1, "tokens: raw string is a single token");
auto raw = std::ranges::find_if(toks, isRaw);
if (raw != toks.end()) {
Check(raw->kind == LintTokenKind::Literal, "tokens: raw string is a Literal");
Check(raw->line == 6, "tokens: raw string starts on line 6");
Check(ctx.TokenText(*raw).contains("// not a comment"), "tokens: raw string body kept intact");
Check(ctx.TokenText(*raw).ends_with(")raw\""), "tokens: raw string spans to its own terminator");
}
// The only comments are the two real ones on lines 4 and 5 — the
// `//` on line 7 lives inside the raw string.
std::vector<std::size_t> commentLines;
for (const LintToken& t : toks) {
if (t.kind == LintTokenKind::Comment) commentLines.push_back(t.line);
}
Check(commentLines == std::vector<std::size_t>{4, 5}, "tokens: only real comments are Comment tokens");
Check(ctx.LineHasComment(4), "tokens: LineHasComment finds a whole-line comment");
Check(ctx.LineHasComment(5), "tokens: LineHasComment finds a trailing comment");
Check(!ctx.LineHasComment(7), "tokens: `//` inside a raw string is not a comment");
Check(!ctx.LineHasComment(2), "tokens: code-only line has no comment");
// TokensOnLine brackets by starting line.
std::span<const LintToken> line2 = ctx.TokensOnLine(2);
Check(!line2.empty() && ctx.TokenText(line2.front()) == "void", "tokens: TokensOnLine starts at the line's first token");
Check(std::ranges::all_of(line2, [](const LintToken& t) { return t.line == 2; }), "tokens: TokensOnLine stays on its line");
// SetContent must invalidate the cache, or offsets point into a
// buffer that no longer exists.
ctx.SetContent("int replaced;\n");
std::span<const LintToken> after = ctx.Tokens();
Check(!after.empty() && ctx.TokenText(after.front()) == "int", "tokens: re-lexed after SetContent");
Check(std::ranges::none_of(after, [&](const LintToken& t) { return ctx.TokenText(t) == "HiddenBranch"; }),
"tokens: stale tokens are dropped after SetContent");
});
RunLint(cfg, Mode(LintMode::Report));
}
// Non-C++ extensions are not lexed: GLSL through a C++ lexer would produce
// plausible-looking nonsense rather than an honest refusal.
{
Scratch s("tokens-foreign");
fs::path shader = s.dir / "f.frag";
{
std::ofstream f(shader, std::ios::binary | std::ios::trunc);
f << "#version 450\nvoid main() { }\n";
}
Configuration cfg = s.Config({});
cfg.shaders.emplace_back(fs::path(shader), "main", ShaderType::Fragment);
cfg.AddLintRule("no-lex", [](LintContext& ctx) {
Check(ctx.Tokens().empty(), "tokens: shaders are not lexed as C++");
});
RunLint(cfg, Mode(LintMode::Report));
}
if (Failures > 0) {
std::println(std::cerr, "{} assertions failed", Failures);
return 1;