diff --git a/implementations/Crafter.Build-Lint.cpp b/implementations/Crafter.Build-Lint.cpp index 45e0d68..93756f6 100644 --- a/implementations/Crafter.Build-Lint.cpp +++ b/implementations/Crafter.Build-Lint.cpp @@ -2,6 +2,12 @@ // SPDX-FileCopyrightText: Copyright (C) 2026 Catcrafts® module; +#include +#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32) +#include +#else +#include +#endif export module Crafter.Build:Lint_impl; import std; import :Lint; @@ -12,6 +18,203 @@ namespace fs = std::filesystem; using namespace Crafter; namespace { + // ---------------- libclang ---------------- + // + // libclang is loaded at runtime rather than linked. Linking -lclang would + // break the mingw and MSVC cross-builds at link time and would put a + // libclang.so.NN runtime dependency into the otherwise self-contained + // release tarballs; the clang-c header is used for its declarations only, + // and every call goes through a pointer resolved here. Failure to load is + // a hard error surfaced once by RunLint — there is deliberately no second, + // weaker lexer to fall back to, because two engines disagreeing about what + // is a comment is a worse failure than not running. +#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32) + using LibHandle = HMODULE; + LibHandle OpenLibrary(const std::string& name) { return LoadLibraryA(name.c_str()); } + void* LibrarySymbol(LibHandle handle, const std::string& name) { return reinterpret_cast(GetProcAddress(handle, name.c_str())); } + constexpr std::string_view LibClangNames[] = {"libclang.dll", "clang.dll"}; +#else + using LibHandle = void*; + LibHandle OpenLibrary(const std::string& name) { return dlopen(name.c_str(), RTLD_NOW | RTLD_LOCAL); } + void* LibrarySymbol(LibHandle handle, const std::string& name) { return dlsym(handle, name.c_str()); } + constexpr std::string_view LibClangNames[] = { + "libclang.so", "libclang.so.22.1", "libclang.so.21.1", "libclang.so.20.1", + "libclang.so.1", "libclang.dylib", + }; +#endif + + // Signatures come from decltype on the header's declarations, so they can + // never drift from the real API. decltype is unevaluated, so naming the + // functions here does not create a link-time reference to them. + struct LibClang { + LibHandle handle = nullptr; + std::string error; // non-empty exactly when handle is null + + decltype(&clang_createIndex) CreateIndex = nullptr; + decltype(&clang_disposeIndex) DisposeIndex = nullptr; + decltype(&clang_parseTranslationUnit) ParseTranslationUnit = nullptr; + decltype(&clang_disposeTranslationUnit) DisposeTranslationUnit = nullptr; + decltype(&clang_getFile) GetFile = nullptr; + decltype(&clang_getLocationForOffset) GetLocationForOffset = nullptr; + decltype(&clang_getRange) GetRange = nullptr; + decltype(&clang_getRangeStart) GetRangeStart = nullptr; + decltype(&clang_getRangeEnd) GetRangeEnd = nullptr; + decltype(&clang_getFileLocation) GetFileLocation = nullptr; + decltype(&clang_tokenize) Tokenize = nullptr; + decltype(&clang_disposeTokens) DisposeTokens = nullptr; + decltype(&clang_getTokenKind) GetTokenKind = nullptr; + decltype(&clang_getTokenExtent) GetTokenExtent = nullptr; + }; + + LibClang LoadLibClang() { + LibClang lib; + std::vector tried; + // CRAFTER_BUILD_LIBCLANG pins an exact path, mirroring the LIBCXX_DIR / + // CRAFTER_MINGW_DIR overrides used elsewhere. It is exclusive: pointing + // it at a broken path must fail loudly rather than quietly succeed with + // some other libclang, or the override is useless for diagnosing which + // library is actually in play. + std::vector candidates; + if (const char* pinned = std::getenv("CRAFTER_BUILD_LIBCLANG"); pinned && *pinned) { + candidates.emplace_back(pinned); + } else { + for (std::string_view name : LibClangNames) candidates.emplace_back(name); + } + + auto join = [](const std::vector& parts) { + std::string joined; + for (const std::string& part : parts) { + if (!joined.empty()) joined += ", "; + joined += part; + } + return joined; + }; + + for (const std::string& name : candidates) { + lib.handle = OpenLibrary(name); + if (lib.handle) break; + tried.push_back(name); + } + if (!lib.handle) { + lib.error = std::format("could not load libclang (tried {}); install clang, or point CRAFTER_BUILD_LIBCLANG at it", join(tried)); + return lib; + } + + std::vector missing; + auto bind = [&](auto& slot, const std::string& name) { + slot = reinterpret_cast>(LibrarySymbol(lib.handle, name)); + if (!slot) missing.push_back(name); + }; + bind(lib.CreateIndex, "clang_createIndex"); + bind(lib.DisposeIndex, "clang_disposeIndex"); + bind(lib.ParseTranslationUnit, "clang_parseTranslationUnit"); + bind(lib.DisposeTranslationUnit, "clang_disposeTranslationUnit"); + bind(lib.GetFile, "clang_getFile"); + bind(lib.GetLocationForOffset, "clang_getLocationForOffset"); + bind(lib.GetRange, "clang_getRange"); + bind(lib.GetRangeStart, "clang_getRangeStart"); + bind(lib.GetRangeEnd, "clang_getRangeEnd"); + bind(lib.GetFileLocation, "clang_getFileLocation"); + bind(lib.Tokenize, "clang_tokenize"); + bind(lib.DisposeTokens, "clang_disposeTokens"); + bind(lib.GetTokenKind, "clang_getTokenKind"); + bind(lib.GetTokenExtent, "clang_getTokenExtent"); + if (!missing.empty()) { + lib.handle = nullptr; + lib.error = std::format("loaded {} but it is missing {}", candidates.front(), join(missing)); + } + return lib; + } + + const LibClang& Clang() { + static const LibClang Lib = LoadLibClang(); + return Lib; + } + + // The -x language for a source file, or empty when we must not lex it. + // .cppm needs c++-module explicitly: libclang does not infer a module unit + // from the extension and silently treats every flag as a linker input if + // left to guess. Shaders and data files return empty — lexing GLSL as C++ + // yields plausible-looking nonsense. + std::string_view LexLanguage(const fs::path& file) { + std::string ext = file.extension().string(); + if (ext == ".cppm" || ext == ".ixx") return "c++-module"; + if (ext == ".cpp" || ext == ".cc" || ext == ".cxx" || ext == ".h" || ext == ".hpp" || ext == ".cu") return "c++"; + if (ext == ".c") return "c"; + return {}; + } + + LintTokenKind MapTokenKind(CXTokenKind kind) { + switch (kind) { + case CXToken_Punctuation: return LintTokenKind::Punctuation; + case CXToken_Keyword: return LintTokenKind::Keyword; + case CXToken_Identifier: return LintTokenKind::Identifier; + case CXToken_Literal: return LintTokenKind::Literal; + case CXToken_Comment: return LintTokenKind::Comment; + } + return LintTokenKind::Punctuation; + } + + // Lex `content` as if it were `file`, returning tokens in source order. + // + // The buffer is handed over as an unsaved file, so a transform's in-memory + // edits are what get lexed — never the stale bytes on disk. The parse is + // expected to fail (a module unit's `import std;` cannot resolve without + // PCMs, and we deliberately do not supply the build's flags here); that + // does not matter, because clang_tokenize re-lexes the buffer and lexing + // has no semantic prerequisites. SingleFileParse keeps it from chasing + // #includes it does not need. + std::vector LexFile(const fs::path& file, const std::string& content) { + std::string_view language = LexLanguage(file); + if (language.empty()) return {}; + const LibClang& lc = Clang(); + if (!lc.handle) return {}; + + std::string path = file.string(); + std::string languageArg = std::format("-x{}", language); + std::string standardArg = language == "c" ? "-std=c23" : "-std=c++26"; + // clang's argv is char* by contract; keep the raw pointers confined to + // this call rather than letting them into any signature of ours. + std::array args{languageArg.c_str(), standardArg.c_str(), "-ferror-limit=0", "-w"}; + + CXUnsavedFile unsaved{}; + unsaved.Filename = path.c_str(); + unsaved.Contents = content.data(); + unsaved.Length = static_cast(content.size()); + + CXIndex index = lc.CreateIndex(0, 0); + if (!index) return {}; + CXTranslationUnit tu = lc.ParseTranslationUnit(index, path.c_str(), args.data(), static_cast(args.size()), &unsaved, 1, CXTranslationUnit_SingleFileParse | CXTranslationUnit_SkipFunctionBodies | CXTranslationUnit_KeepGoing); + if (!tu) { + lc.DisposeIndex(index); + return {}; + } + + std::vector tokens; + if (CXFile cxFile = lc.GetFile(tu, path.c_str())) { + CXSourceRange whole = lc.GetRange(lc.GetLocationForOffset(tu, cxFile, 0), lc.GetLocationForOffset(tu, cxFile, static_cast(content.size()))); + CXToken* raw = nullptr; + std::uint32_t count = 0; + lc.Tokenize(tu, whole, &raw, &count); + tokens.reserve(count); + for (std::uint32_t i = 0; i < count; ++i) { + CXSourceRange extent = lc.GetTokenExtent(tu, raw[i]); + std::uint32_t line = 0; + std::uint32_t column = 0; + std::uint32_t begin = 0; + std::uint32_t end = 0; + lc.GetFileLocation(lc.GetRangeStart(extent), nullptr, &line, &column, &begin); + lc.GetFileLocation(lc.GetRangeEnd(extent), nullptr, nullptr, nullptr, &end); + if (end < begin || begin > content.size()) continue; + tokens.push_back({MapTokenKind(lc.GetTokenKind(raw[i])), begin, std::min(end - begin, content.size() - begin), line, column}); + } + if (raw) lc.DisposeTokens(tu, raw, count); + } + lc.DisposeTranslationUnit(tu); + lc.DisposeIndex(index); + return tokens; + } + // Blank //-comments, /*...*/ comments and string/char literal bodies to // spaces while copying '\n' through, so byte offsets and line numbers in // the result match the original text. Raw string literals are not @@ -164,6 +367,30 @@ const std::string& LintContext::CommentStripped() { return *commentStrippedCache; } +std::span LintContext::Tokens() { + if (!tokenCache) tokenCache = LexFile(file, content); + return *tokenCache; +} + +std::string_view LintContext::TokenText(const LintToken& token) const { + if (token.offset >= content.size()) return {}; + return std::string_view(content).substr(token.offset, token.length); +} + +std::span LintContext::TokensOnLine(std::size_t line) { + // Tokens come back in source order, so one line's tokens are a contiguous + // run and can be bracketed by binary search. + std::span all = Tokens(); + auto begin = std::ranges::lower_bound(all, line, {}, &LintToken::line); + auto end = std::ranges::upper_bound(all, line, {}, &LintToken::line); + return all.subspan(static_cast(begin - all.begin()), static_cast(end - begin)); +} + +bool LintContext::LineHasComment(std::size_t line) { + std::span onLine = TokensOnLine(line); + return std::ranges::any_of(onLine, [](const LintToken& t) { return t.kind == LintTokenKind::Comment; }); +} + void LintContext::Report(std::size_t line, std::string message) { sink->push_back({file, line, activeRule, std::move(message)}); } @@ -226,6 +453,7 @@ void LintContext::SetContent(std::string newContent) { lines = SplitLines(content); commentStrippedCache.reset(); suppressionsCache.reset(); // line numbers may have shifted — re-parse + tokenCache.reset(); // offsets refer to the old buffer — re-lex } void Configuration::AddLintRule(std::string name, std::function check) { @@ -235,6 +463,15 @@ void Configuration::AddLintRule(std::string name, std::function> lineRules; }; + // Lexical class of a LintToken, mirroring clang's token kinds one-to-one. + enum class LintTokenKind { + Punctuation, + Keyword, + Identifier, + Literal, // string, raw string, character, integer, floating literal + Comment, // // ... or /* ... */ + }; + + // One token from LintContext::Tokens(). `offset`/`length` are byte offsets + // into LintContext::content and stay valid until the next SetContent. + // + // A raw string literal or a block comment is ONE token and may span lines, + // which is exactly what the hand-rolled scanners could not represent. The + // stream also covers text inside preprocessor branches that are inactive + // for the host — clang_tokenize lexes, it does not evaluate #if — so token + // rules see every platform's code, not just the one being built. + struct LintToken { + LintTokenKind kind = LintTokenKind::Punctuation; + std::size_t offset = 0; + std::size_t length = 0; + std::size_t line = 0; // 1-based, of the token's first byte + std::size_t column = 0; // 1-based, of the token's first byte + }; + // Per-file view handed to each LintRule's check callback. Every member // function is out-of-line and CRAFTER_API (defined in Crafter.Build:Lint's // implementation unit) because rule lambdas execute from the user's @@ -165,11 +190,31 @@ export namespace Crafter { // re-parsed after SetContent. CRAFTER_API bool Suppressed(std::string_view rule, std::size_t line); + // The file lexed by clang, in source order. Built on first call, cached + // per file, re-lexed after SetContent. Empty for extensions that are + // not C or C++ (shaders): lexing GLSL as C++ would produce nonsense. + // + // Prefer this to scanning characters. It is the only view that gets + // raw strings, line splices, digraphs and nested quoting right, and + // offsets index straight into `content`, so a transform can find in + // the token stream and edit in place. + CRAFTER_API std::span Tokens(); + // The token's own bytes: content.substr(tok.offset, tok.length). + CRAFTER_API std::string_view TokenText(const LintToken& token) const; + // Tokens whose FIRST byte is on `line` (1-based). A token that starts + // earlier and spans into `line` — a raw string, a block comment — is + // not included; ask Tokens() directly when that matters. + CRAFTER_API std::span TokensOnLine(std::size_t line); + // True when a comment token starts on `line`. Replaces `Line(n).contains("//")`, + // which false-positives on a `//` inside a string literal. + CRAFTER_API bool LineHasComment(std::size_t line); + // Driver wiring — set by RunLint before each check call. Not for rules. std::string activeRule; std::vector* sink = nullptr; std::optional commentStrippedCache; std::optional suppressionsCache; + std::optional> tokenCache; }; // A named lint rule: `check` runs once per (rule, file) over the project's diff --git a/interfaces/Crafter.Build-Lint.cppm b/interfaces/Crafter.Build-Lint.cppm index d6c2a8a..7d45b3f 100644 --- a/interfaces/Crafter.Build-Lint.cppm +++ b/interfaces/Crafter.Build-Lint.cppm @@ -39,10 +39,13 @@ export namespace Crafter { std::vector changedFiles; std::size_t filesLinted = 0; std::size_t rulesRun = 0; // rules remaining after glob filter - std::size_t errors = 0; // rule exceptions + write failures + // Rule exceptions, write failures, and infrastructure failures such as + // libclang not loading. Counted in Clean() so a run that could not do + // its job never looks like a run that found nothing. + std::size_t errors = 0; bool noRulesDefined = false; // project registered no rules at all // Host-side only (like TestSummary::AllPassed), safe as in-class inline. - bool Clean() const { return findings.empty() && !noRulesDefined; } + bool Clean() const { return findings.empty() && !noRulesDefined && errors == 0; } }; // Run the project's lint rules over its own sources: module interfaces diff --git a/tests/Lint/main.cpp b/tests/Lint/main.cpp index c900b27..1af4d0c 100644 --- a/tests/Lint/main.cpp +++ b/tests/Lint/main.cpp @@ -427,6 +427,106 @@ int main() { Check(s.Read("a") == "// lint-disable-file flag\nbad\n", "other rules still format"); } + // ---------------- token layer ---------------- + // + // The source is spelled with escaped literals rather than a raw string so + // that this file stays lintable by the very rules under test; the scratch + // file it writes does contain a genuine multi-line raw string. + { + constexpr std::string_view Source = + "#ifdef CRAFTER_LINT_NEVER_DEFINED\n" // 1 + "void HiddenBranch();\n" // 2 + "#endif\n" // 3 + "// a real comment\n" // 4 + "int url = 1; // https://example.com\n" // 5 + "auto raw = R\"raw(spans lines\n" // 6 + " // not a comment\n" // 7 + " int notADecl;\n" // 8 + ")raw\";\n"; // 9 + + Scratch s("tokens"); + s.Write("f", Source); + Configuration cfg = s.Config({"f"}); + cfg.AddLintRule("tokens", [](LintContext& ctx) { + std::span toks = ctx.Tokens(); + Check(!toks.empty(), "tokens: file lexes to a non-empty stream"); + + // Every token's offset/length must address its own bytes, or a + // transform editing at an offset would corrupt the file. + bool offsetsSound = true; + for (const LintToken& t : toks) { + if (t.offset + t.length > ctx.content.size() || ctx.TokenText(t).empty()) offsetsSound = false; + } + Check(offsetsSound, "tokens: every offset/length addresses real bytes"); + + // Ordered by offset, so binary search in TokensOnLine is valid. + bool ordered = std::ranges::is_sorted(toks, {}, &LintToken::offset); + Check(ordered, "tokens: stream is in source order"); + + // Inactive #ifdef branch is still lexed — this is what keeps token + // rules covering every platform, unlike an AST. + bool sawHidden = std::ranges::any_of(toks, [&](const LintToken& t) { + return t.kind == LintTokenKind::Identifier && ctx.TokenText(t) == "HiddenBranch"; + }); + Check(sawHidden, "tokens: inactive #ifdef branch is lexed"); + + // A multi-line raw string is exactly one Literal, comment markers + // and declarations inside it included. + auto isRaw = [&](const LintToken& t) { return ctx.TokenText(t).starts_with("R\"raw("); }; + Check(std::ranges::count_if(toks, isRaw) == 1, "tokens: raw string is a single token"); + auto raw = std::ranges::find_if(toks, isRaw); + if (raw != toks.end()) { + Check(raw->kind == LintTokenKind::Literal, "tokens: raw string is a Literal"); + Check(raw->line == 6, "tokens: raw string starts on line 6"); + Check(ctx.TokenText(*raw).contains("// not a comment"), "tokens: raw string body kept intact"); + Check(ctx.TokenText(*raw).ends_with(")raw\""), "tokens: raw string spans to its own terminator"); + } + + // The only comments are the two real ones on lines 4 and 5 — the + // `//` on line 7 lives inside the raw string. + std::vector commentLines; + for (const LintToken& t : toks) { + if (t.kind == LintTokenKind::Comment) commentLines.push_back(t.line); + } + Check(commentLines == std::vector{4, 5}, "tokens: only real comments are Comment tokens"); + Check(ctx.LineHasComment(4), "tokens: LineHasComment finds a whole-line comment"); + Check(ctx.LineHasComment(5), "tokens: LineHasComment finds a trailing comment"); + Check(!ctx.LineHasComment(7), "tokens: `//` inside a raw string is not a comment"); + Check(!ctx.LineHasComment(2), "tokens: code-only line has no comment"); + + // TokensOnLine brackets by starting line. + std::span line2 = ctx.TokensOnLine(2); + Check(!line2.empty() && ctx.TokenText(line2.front()) == "void", "tokens: TokensOnLine starts at the line's first token"); + Check(std::ranges::all_of(line2, [](const LintToken& t) { return t.line == 2; }), "tokens: TokensOnLine stays on its line"); + + // SetContent must invalidate the cache, or offsets point into a + // buffer that no longer exists. + ctx.SetContent("int replaced;\n"); + std::span after = ctx.Tokens(); + Check(!after.empty() && ctx.TokenText(after.front()) == "int", "tokens: re-lexed after SetContent"); + Check(std::ranges::none_of(after, [&](const LintToken& t) { return ctx.TokenText(t) == "HiddenBranch"; }), + "tokens: stale tokens are dropped after SetContent"); + }); + RunLint(cfg, Mode(LintMode::Report)); + } + + // Non-C++ extensions are not lexed: GLSL through a C++ lexer would produce + // plausible-looking nonsense rather than an honest refusal. + { + Scratch s("tokens-foreign"); + fs::path shader = s.dir / "f.frag"; + { + std::ofstream f(shader, std::ios::binary | std::ios::trunc); + f << "#version 450\nvoid main() { }\n"; + } + Configuration cfg = s.Config({}); + cfg.shaders.emplace_back(fs::path(shader), "main", ShaderType::Fragment); + cfg.AddLintRule("no-lex", [](LintContext& ctx) { + Check(ctx.Tokens().empty(), "tokens: shaders are not lexed as C++"); + }); + RunLint(cfg, Mode(LintMode::Report)); + } + if (Failures > 0) { std::println(std::cerr, "{} assertions failed", Failures); return 1;