diff --git a/implementations/Crafter.Build-Clang.cpp b/implementations/Crafter.Build-Clang.cpp index 3599f3c..28491f8 100644 --- a/implementations/Crafter.Build-Clang.cpp +++ b/implementations/Crafter.Build-Clang.cpp @@ -1801,21 +1801,6 @@ int Crafter::Run(int argc, char** argv) { " header Cross-Origin-Embedder-Policy \"require-corp\"\n" " header Cross-Origin-Resource-Policy \"same-origin\"\n" " header Cache-Control \"no-store\"\n" - // Every EnableWasiBrowserRuntime consumer is a - // single-page wasm app by construction: the - // generated index.html ships an empty and - // the module builds the DOM at runtime. So an app - // that routes on window.location has no file on - // disk for any path but "/", and a bare - // file_server 404s every deep link, refresh and - // shared URL during development. - // - // try_files falls through to index.html only for - // paths that are not real files, so static assets - // still serve normally and a genuinely missing - // asset becomes a visible wrong-content-type - // rather than a silent 404 the app can't see. - " try_files {{path}} /index.html\n" " file_server\n" "}}\n", port, absDir.string())); diff --git a/implementations/Crafter.Build-Lint.cpp b/implementations/Crafter.Build-Lint.cpp index 8571e29..45e0d68 100644 --- a/implementations/Crafter.Build-Lint.cpp +++ b/implementations/Crafter.Build-Lint.cpp @@ -2,12 +2,6 @@ // SPDX-FileCopyrightText: Copyright (C) 2026 Catcrafts® module; -#include -#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32) -#include -#else -#include -#endif export module Crafter.Build:Lint_impl; import std; import :Lint; @@ -18,242 +12,72 @@ namespace fs = std::filesystem; using namespace Crafter; namespace { - // ---------------- libclang ---------------- - // - // libclang is loaded at runtime rather than linked. Linking -lclang would - // break the mingw and MSVC cross-builds at link time and would put a - // libclang.so.NN runtime dependency into the otherwise self-contained - // release tarballs; the clang-c header is used for its declarations only, - // and every call goes through a pointer resolved here. Failure to load is - // a hard error surfaced once by RunLint — there is deliberately no second, - // weaker lexer to fall back to, because two engines disagreeing about what - // is a comment is a worse failure than not running. -#if defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_pc_windows_msvc) || defined(CRAFTER_BUILD_CONFIGURATION_TARGET_x86_64_w64_mingw32) - using LibHandle = HMODULE; - LibHandle OpenLibrary(const std::string& name) { return LoadLibraryA(name.c_str()); } - void* LibrarySymbol(LibHandle handle, const std::string& name) { return reinterpret_cast(GetProcAddress(handle, name.c_str())); } - constexpr std::string_view LibClangNames[] = {"libclang.dll", "clang.dll"}; -#else - using LibHandle = void*; - LibHandle OpenLibrary(const std::string& name) { return dlopen(name.c_str(), RTLD_NOW | RTLD_LOCAL); } - void* LibrarySymbol(LibHandle handle, const std::string& name) { return dlsym(handle, name.c_str()); } - constexpr std::string_view LibClangNames[] = { - "libclang.so", "libclang.so.22.1", "libclang.so.21.1", "libclang.so.20.1", - "libclang.so.1", "libclang.dylib", - }; -#endif - - // Signatures come from decltype on the header's declarations, so they can - // never drift from the real API. decltype is unevaluated, so naming the - // functions here does not create a link-time reference to them. - struct LibClang { - LibHandle handle = nullptr; - std::string error; // non-empty exactly when handle is null - - decltype(&clang_createIndex) CreateIndex = nullptr; - decltype(&clang_disposeIndex) DisposeIndex = nullptr; - decltype(&clang_parseTranslationUnit) ParseTranslationUnit = nullptr; - decltype(&clang_disposeTranslationUnit) DisposeTranslationUnit = nullptr; - decltype(&clang_getFile) GetFile = nullptr; - decltype(&clang_getLocationForOffset) GetLocationForOffset = nullptr; - decltype(&clang_getRange) GetRange = nullptr; - decltype(&clang_getRangeStart) GetRangeStart = nullptr; - decltype(&clang_getRangeEnd) GetRangeEnd = nullptr; - decltype(&clang_getFileLocation) GetFileLocation = nullptr; - decltype(&clang_tokenize) Tokenize = nullptr; - decltype(&clang_disposeTokens) DisposeTokens = nullptr; - decltype(&clang_getTokenKind) GetTokenKind = nullptr; - decltype(&clang_getTokenExtent) GetTokenExtent = nullptr; - }; - - LibClang LoadLibClang() { - LibClang lib; - std::vector tried; - // CRAFTER_BUILD_LIBCLANG pins an exact path, mirroring the LIBCXX_DIR / - // CRAFTER_MINGW_DIR overrides used elsewhere. It is exclusive: pointing - // it at a broken path must fail loudly rather than quietly succeed with - // some other libclang, or the override is useless for diagnosing which - // library is actually in play. - std::vector candidates; - if (const char* pinned = std::getenv("CRAFTER_BUILD_LIBCLANG"); pinned && *pinned) { - candidates.emplace_back(pinned); - } else { - for (std::string_view name : LibClangNames) candidates.emplace_back(name); - } - - auto join = [](const std::vector& parts) { - std::string joined; - for (const std::string& part : parts) { - if (!joined.empty()) joined += ", "; - joined += part; + // Blank //-comments, /*...*/ comments and string/char literal bodies to + // spaces while copying '\n' through, so byte offsets and line numbers in + // the result match the original text. Raw string literals are not + // recognized (documented v1 limitation) — their bodies pass through as + // ordinary string content until the first '"'. + std::string StripComments(std::string_view src) { + enum class State { Code, LineComment, BlockComment, String, Char }; + std::string out; + out.reserve(src.size()); + State state = State::Code; + for (std::size_t i = 0; i < src.size(); ++i) { + char c = src[i]; + char next = i + 1 < src.size() ? src[i + 1] : '\0'; + switch (state) { + case State::Code: + if (c == '/' && next == '/') { + state = State::LineComment; + out += " "; + ++i; + } else if (c == '/' && next == '*') { + state = State::BlockComment; + out += " "; + ++i; + } else if (c == '"') { + state = State::String; + out += c; // keep the delimiter so quoting stays visible + } else if (c == '\'') { + state = State::Char; + out += c; + } else { + out += c; + } + break; + case State::LineComment: + if (c == '\n') { + state = State::Code; + out += c; + } else { + out += ' '; + } + break; + case State::BlockComment: + if (c == '*' && next == '/') { + state = State::Code; + out += " "; + ++i; + } else { + out += c == '\n' ? '\n' : ' '; + } + break; + case State::String: + case State::Char: { + char delim = state == State::String ? '"' : '\''; + if (c == '\\' && next != '\0') { + out += " "; + ++i; + if (next == '\n') out.back() = '\n'; + } else if (c == delim) { + state = State::Code; + out += c; + } else { + out += c == '\n' ? '\n' : ' '; + } + break; + } } - return joined; - }; - - for (const std::string& name : candidates) { - lib.handle = OpenLibrary(name); - if (lib.handle) break; - tried.push_back(name); - } - if (!lib.handle) { - lib.error = std::format("could not load libclang (tried {}); install clang, or point CRAFTER_BUILD_LIBCLANG at it", join(tried)); - return lib; - } - - std::vector missing; - auto bind = [&](auto& slot, const std::string& name) { - slot = reinterpret_cast>(LibrarySymbol(lib.handle, name)); - if (!slot) missing.push_back(name); - }; - bind(lib.CreateIndex, "clang_createIndex"); - bind(lib.DisposeIndex, "clang_disposeIndex"); - bind(lib.ParseTranslationUnit, "clang_parseTranslationUnit"); - bind(lib.DisposeTranslationUnit, "clang_disposeTranslationUnit"); - bind(lib.GetFile, "clang_getFile"); - bind(lib.GetLocationForOffset, "clang_getLocationForOffset"); - bind(lib.GetRange, "clang_getRange"); - bind(lib.GetRangeStart, "clang_getRangeStart"); - bind(lib.GetRangeEnd, "clang_getRangeEnd"); - bind(lib.GetFileLocation, "clang_getFileLocation"); - bind(lib.Tokenize, "clang_tokenize"); - bind(lib.DisposeTokens, "clang_disposeTokens"); - bind(lib.GetTokenKind, "clang_getTokenKind"); - bind(lib.GetTokenExtent, "clang_getTokenExtent"); - if (!missing.empty()) { - lib.handle = nullptr; - lib.error = std::format("loaded {} but it is missing {}", candidates.front(), join(missing)); - } - return lib; - } - - const LibClang& Clang() { - static const LibClang Lib = LoadLibClang(); - return Lib; - } - - // The -x language for a source file, or empty when we must not lex it. - // .cppm needs c++-module explicitly: libclang does not infer a module unit - // from the extension and silently treats every flag as a linker input if - // left to guess. Shaders and data files return empty — lexing GLSL as C++ - // yields plausible-looking nonsense. - std::string_view LexLanguage(const fs::path& file) { - std::string ext = file.extension().string(); - if (ext == ".cppm" || ext == ".ixx") return "c++-module"; - if (ext == ".cpp" || ext == ".cc" || ext == ".cxx" || ext == ".h" || ext == ".hpp" || ext == ".cu") return "c++"; - if (ext == ".c") return "c"; - return {}; - } - - LintTokenKind MapTokenKind(CXTokenKind kind) { - switch (kind) { - case CXToken_Punctuation: return LintTokenKind::Punctuation; - case CXToken_Keyword: return LintTokenKind::Keyword; - case CXToken_Identifier: return LintTokenKind::Identifier; - case CXToken_Literal: return LintTokenKind::Literal; - case CXToken_Comment: return LintTokenKind::Comment; - } - return LintTokenKind::Punctuation; - } - - // Lex `content` as if it were `file`, returning tokens in source order. - // - // The buffer is handed over as an unsaved file, so a transform's in-memory - // edits are what get lexed — never the stale bytes on disk. The parse is - // expected to fail (a module unit's `import std;` cannot resolve without - // PCMs, and we deliberately do not supply the build's flags here); that - // does not matter, because clang_tokenize re-lexes the buffer and lexing - // has no semantic prerequisites. SingleFileParse keeps it from chasing - // #includes it does not need. - std::vector LexFile(const fs::path& file, const std::string& content) { - std::string_view language = LexLanguage(file); - if (language.empty()) return {}; - const LibClang& lc = Clang(); - if (!lc.handle) return {}; - - std::string path = file.string(); - std::string languageArg = std::format("-x{}", language); - std::string standardArg = language == "c" ? "-std=c23" : "-std=c++26"; - // clang's argv is char* by contract; keep the raw pointers confined to - // this call rather than letting them into any signature of ours. - std::array args{languageArg.c_str(), standardArg.c_str(), "-ferror-limit=0", "-w"}; - - CXUnsavedFile unsaved{}; - unsaved.Filename = path.c_str(); - unsaved.Contents = content.data(); - unsaved.Length = static_cast(content.size()); - - CXIndex index = lc.CreateIndex(0, 0); - if (!index) return {}; - CXTranslationUnit tu = lc.ParseTranslationUnit(index, path.c_str(), args.data(), static_cast(args.size()), &unsaved, 1, CXTranslationUnit_SingleFileParse | CXTranslationUnit_SkipFunctionBodies | CXTranslationUnit_KeepGoing); - if (!tu) { - lc.DisposeIndex(index); - return {}; - } - - std::vector tokens; - if (CXFile cxFile = lc.GetFile(tu, path.c_str())) { - CXSourceRange whole = lc.GetRange(lc.GetLocationForOffset(tu, cxFile, 0), lc.GetLocationForOffset(tu, cxFile, static_cast(content.size()))); - CXToken* raw = nullptr; - std::uint32_t count = 0; - lc.Tokenize(tu, whole, &raw, &count); - tokens.reserve(count); - for (std::uint32_t i = 0; i < count; ++i) { - CXSourceRange extent = lc.GetTokenExtent(tu, raw[i]); - std::uint32_t line = 0; - std::uint32_t column = 0; - std::uint32_t begin = 0; - std::uint32_t end = 0; - lc.GetFileLocation(lc.GetRangeStart(extent), nullptr, &line, &column, &begin); - lc.GetFileLocation(lc.GetRangeEnd(extent), nullptr, nullptr, nullptr, &end); - if (end < begin || begin > content.size()) continue; - tokens.push_back({MapTokenKind(lc.GetTokenKind(raw[i])), begin, std::min(end - begin, content.size() - begin), line, column}); - } - if (raw) lc.DisposeTokens(tu, raw, count); - } - lc.DisposeTranslationUnit(tu); - lc.DisposeIndex(index); - return tokens; - } - - // Blank comments and the bodies of string/character literals to spaces, - // copying '\n' through so byte offsets and line numbers in the result - // match the original text exactly. - // - // Derived from the token stream rather than scanned character by - // character, which is what makes raw strings, escapes, encoding prefixes - // and a literal like '"' come out right. Editing a copy of the buffer in - // place — rather than appending to a fresh string — makes the - // length-preserving property structural instead of something every branch - // has to remember. - std::string StripLiterals(const std::string& content, std::span tokens) { - std::string out = content; - auto blank = [&out](std::size_t from, std::size_t to) { - for (std::size_t i = from; i < to && i < out.size(); ++i) { - if (out[i] != '\n') out[i] = ' '; - } - }; - for (const LintToken& token : tokens) { - std::size_t begin = token.offset; - std::size_t end = token.offset + token.length; - if (token.kind == LintTokenKind::Comment) { - blank(begin, end); - continue; - } - if (token.kind != LintTokenKind::Literal || token.length < 2) continue; - std::string_view text(content.data() + begin, token.length); - // Numeric literals are code and stay; only string and character - // literals have a body to hide. A digit separator makes 1'000 look - // quote-ish, so require everything before the quote to be an - // encoding prefix (L, u, U, u8, R and their combinations). - std::size_t quote = text.find_first_of("\"'"); - if (quote == std::string_view::npos) continue; - std::string_view prefix = text.substr(0, quote); - if (!std::ranges::all_of(prefix, [](char c) { return c == 'L' || c == 'u' || c == 'U' || c == '8' || c == 'R'; })) continue; - // Keep the opening quote and the closing one, blank everything - // between. For a raw string that also blanks the R"delim( and - // )delim" scaffolding, leaving exactly two quotes — which is what - // rules counting quotes to find a literal's extent rely on. - blank(begin + quote + 1, end - 1); } return out; } @@ -335,50 +159,11 @@ std::string_view LintContext::Line(std::size_t n) const { const std::string& LintContext::CommentStripped() { if (!commentStrippedCache) { - commentStrippedCache = StripLiterals(content, Tokens()); + commentStrippedCache = StripComments(content); } return *commentStrippedCache; } -std::span LintContext::Tokens() { - if (!tokenCache) tokenCache = LexFile(file, content); - return *tokenCache; -} - -std::string_view LintContext::TokenText(const LintToken& token) const { - if (token.offset >= content.size()) return {}; - return std::string_view(content).substr(token.offset, token.length); -} - -std::span LintContext::TokensOnLine(std::size_t line) { - // Tokens come back in source order, so one line's tokens are a contiguous - // run and can be bracketed by binary search. - std::span all = Tokens(); - auto begin = std::ranges::lower_bound(all, line, {}, &LintToken::line); - auto end = std::ranges::upper_bound(all, line, {}, &LintToken::line); - return all.subspan(static_cast(begin - all.begin()), static_cast(end - begin)); -} - -bool LintContext::LineHasComment(std::size_t line) { - std::span onLine = TokensOnLine(line); - return std::ranges::any_of(onLine, [](const LintToken& t) { return t.kind == LintTokenKind::Comment; }); -} - -bool LintContext::LineHasMultiLineToken(std::size_t line) { - if (!spannedLineCache) { - std::vector spanned(lines.size(), false); - for (const LintToken& token : Tokens()) { - std::size_t crossed = static_cast(std::ranges::count(TokenText(token), '\n')); - if (crossed == 0) continue; - for (std::size_t n = token.line; n <= token.line + crossed && n <= spanned.size(); ++n) { - spanned[n - 1] = true; - } - } - spannedLineCache = std::move(spanned); - } - return line >= 1 && line <= spannedLineCache->size() && (*spannedLineCache)[line - 1]; -} - void LintContext::Report(std::size_t line, std::string message) { sink->push_back({file, line, activeRule, std::move(message)}); } @@ -441,8 +226,6 @@ void LintContext::SetContent(std::string newContent) { lines = SplitLines(content); commentStrippedCache.reset(); suppressionsCache.reset(); // line numbers may have shifted — re-parse - tokenCache.reset(); // offsets refer to the old buffer — re-lex - spannedLineCache.reset(); // derived from tokenCache } void Configuration::AddLintRule(std::string name, std::function check) { @@ -452,15 +235,6 @@ void Configuration::AddLintRule(std::string name, std::function> lineRules; }; - // Lexical class of a LintToken, mirroring clang's token kinds one-to-one. - enum class LintTokenKind { - Punctuation, - Keyword, - Identifier, - Literal, // string, raw string, character, integer, floating literal - Comment, // // ... or /* ... */ - }; - - // One token from LintContext::Tokens(). `offset`/`length` are byte offsets - // into LintContext::content and stay valid until the next SetContent. - // - // A raw string literal or a block comment is ONE token and may span lines, - // which is exactly what the hand-rolled scanners could not represent. The - // stream also covers text inside preprocessor branches that are inactive - // for the host — clang_tokenize lexes, it does not evaluate #if — so token - // rules see every platform's code, not just the one being built. - struct LintToken { - LintTokenKind kind = LintTokenKind::Punctuation; - std::size_t offset = 0; - std::size_t length = 0; - std::size_t line = 0; // 1-based, of the token's first byte - std::size_t column = 0; // 1-based, of the token's first byte - }; - // Per-file view handed to each LintRule's check callback. Every member // function is out-of-line and CRAFTER_API (defined in Crafter.Build:Lint's // implementation unit) because rule lambdas execute from the user's @@ -161,16 +136,10 @@ export namespace Crafter { CRAFTER_API std::string Extension() const; // ".cppm", ".cpp", ".h", ... CRAFTER_API std::string_view Line(std::size_t n) const; // 1-based; empty if out of range - // `content` with comments and string/char literal bodies blanked to - // spaces, newlines preserved — offsets and line numbers stay valid. - // Built from Tokens() on first call, cached per file, so raw strings, - // escapes, encoding prefixes and literals like '"' all come out right. - // A raw string is reduced to R"…" with the body and the delimiter - // scaffolding blanked, leaving exactly two quote characters. - // - // Convenient for a quick scan, but Tokens() is the better tool for - // anything structural: this view cannot tell an identifier from a - // keyword, and it has already thrown away where the literals were. + // `content` with //-comments, /*...*/ comments and string/char literal + // bodies blanked to spaces, newlines preserved — offsets and line + // numbers stay valid. Built on first call, cached per file. Raw string + // literals are not recognized (v1 limitation). CRAFTER_API const std::string& CommentStripped(); // Record a finding at `line` (1-based; pass 0 for a whole-file finding). CRAFTER_API void Report(std::size_t line, std::string message); @@ -196,42 +165,11 @@ export namespace Crafter { // re-parsed after SetContent. CRAFTER_API bool Suppressed(std::string_view rule, std::size_t line); - // The file lexed by clang, in source order. Built on first call, cached - // per file, re-lexed after SetContent. Empty for extensions that are - // not C or C++ (shaders): lexing GLSL as C++ would produce nonsense. - // - // Prefer this to scanning characters. It is the only view that gets - // raw strings, line splices, digraphs and nested quoting right, and - // offsets index straight into `content`, so a transform can find in - // the token stream and edit in place. - CRAFTER_API std::span Tokens(); - // The token's own bytes: content.substr(tok.offset, tok.length). - CRAFTER_API std::string_view TokenText(const LintToken& token) const; - // Tokens whose FIRST byte is on `line` (1-based). A token that starts - // earlier and spans into `line` — a raw string, a block comment — is - // not included; ask Tokens() directly when that matters. - CRAFTER_API std::span TokensOnLine(std::size_t line); - // True when a comment token starts on `line`. Replaces `Line(n).contains("//")`, - // which false-positives on a `//` inside a string literal. - CRAFTER_API bool LineHasComment(std::size_t line); - // True when `line` (1-based) is touched by a token that spans more - // than one line — a raw string or a block comment. A transform that - // joins, splits or rewrites such a line changes what is inside that - // token, so this is the guard to consult before reflowing anything. - // Replaces `Line(n).contains("R\"")`, which both misses raw strings - // opened on an earlier line and fires on the characters R" appearing - // inside an ordinary literal. - CRAFTER_API bool LineHasMultiLineToken(std::size_t line); - // Driver wiring — set by RunLint before each check call. Not for rules. std::string activeRule; std::vector* sink = nullptr; std::optional commentStrippedCache; std::optional suppressionsCache; - std::optional> tokenCache; - // Per-line flag, 0-based, for LineHasMultiLineToken. Derived from - // tokenCache and invalidated with it. - std::optional> spannedLineCache; }; // A named lint rule: `check` runs once per (rule, file) over the project's diff --git a/interfaces/Crafter.Build-Lint.cppm b/interfaces/Crafter.Build-Lint.cppm index 7d45b3f..d6c2a8a 100644 --- a/interfaces/Crafter.Build-Lint.cppm +++ b/interfaces/Crafter.Build-Lint.cppm @@ -39,13 +39,10 @@ export namespace Crafter { std::vector changedFiles; std::size_t filesLinted = 0; std::size_t rulesRun = 0; // rules remaining after glob filter - // Rule exceptions, write failures, and infrastructure failures such as - // libclang not loading. Counted in Clean() so a run that could not do - // its job never looks like a run that found nothing. - std::size_t errors = 0; + std::size_t errors = 0; // rule exceptions + write failures bool noRulesDefined = false; // project registered no rules at all // Host-side only (like TestSummary::AllPassed), safe as in-class inline. - bool Clean() const { return findings.empty() && !noRulesDefined && errors == 0; } + bool Clean() const { return findings.empty() && !noRulesDefined; } }; // Run the project's lint rules over its own sources: module interfaces diff --git a/lint-rules.h b/lint-rules.h index 3cf8c92..12461fd 100644 --- a/lint-rules.h +++ b/lint-rules.h @@ -408,12 +408,9 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { lineStart += line.size() + 1; std::string_view trimmed = Trim(line); if (trimmed.starts_with('#')) continue; - // A chain is read within one line, so a literal that spans lines - // would be sliced in half by chainEnd. Single-line raw strings are - // fine: they reduce to R"…" in the stripped view, so they fail the - // "is a plain literal" test below and travel through as an - // argument, spelling and all. - bool spansLines = ctx.LineHasMultiLineToken(li + 1); + // Raw strings defeat the comment stripper's quote tracking; any + // literal-+ pattern on such a line is a manual fix. + bool hasRaw = std::string_view(ctx.content).substr(lineOff, line.size()).contains("R\""); std::size_t searchFrom = 0; while (searchFrom < line.size()) { @@ -430,8 +427,8 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { bool literalAdjacent = (leftEnd > 0 && line[leftEnd - 1] == '"') || (rightBegin < line.size() && line[rightBegin] == '"'); if (!literalAdjacent) continue; - if (spansLines) { - ctx.Report(li + 1, "use std::format instead of string concatenation with + (multi-line literal, fix manually)"); + if (hasRaw) { + ctx.Report(li + 1, "use std::format instead of string concatenation with + (raw-string line, fix manually)"); break; } @@ -610,7 +607,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { std::string lineStr(stripped[i]); std::smatch m; std::string_view raw = i + 1 <= ctx.lines.size() ? ctx.Line(i + 1) : std::string_view{}; - if (!ctx.LineHasComment(i + 1) && !ctx.Suppressed("single-declaration", i + 1) + if (!raw.contains("//") && !ctx.Suppressed("single-declaration", i + 1) && std::regex_match(lineStr, m, simpleMulti)) { std::string indent = m[1].str(), type = m[2].str(), decls = m[3].str(); std::size_t start = 0; @@ -647,7 +644,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { if (trimmed == "{" && !outLines.empty()) { std::string_view prevTrim = i > 0 ? Trim(stripped[i - 1]) : std::string_view{}; bool headerBefore = prevTrim.ends_with(')') || prevTrim == "else" || prevTrim == "do" || prevTrim == "try"; - bool hasComment = ctx.LineHasComment(i + 1) || (i > 0 && ctx.LineHasComment(i)); + bool hasComment = ctx.Line(i + 1).contains("//") || (i > 0 && ctx.Line(i).contains("//")); bool suppressed = ctx.Suppressed("brace-style", i) || ctx.Suppressed("brace-style", i + 1); if (headerBefore && !hasComment && !suppressed) { std::string& prev = outLines.back(); @@ -683,7 +680,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { if (i + 1 < ctx.lines.size() && std::regex_match(lineStr, ifHeader) && ParenDelta(stripped[i]) == 0) { std::string_view body = Trim(stripped[i + 1]); bool joinable = !body.empty() && body != "{" && !body.starts_with("if") && body.ends_with(';') - && !ctx.LineHasComment(i + 1) && !ctx.LineHasComment(i + 2) + && !ctx.Line(i + 1).contains("//") && !ctx.Line(i + 2).contains("//") && !ctx.Suppressed("if-single-line", i + 1) && !ctx.Suppressed("if-single-line", i + 2); std::string joined = std::string(ctx.Line(i + 1)); while (!joined.empty() && (joined.back() == ' ' || joined.back() == '\t')) joined.pop_back(); @@ -727,7 +724,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { std::string_view trimmed = Trim(stripped[i]); std::int64_t delta = ParenDelta(stripped[i]); bool candidate = delta > 0 && !trimmed.empty() && !trimmed.starts_with('#') - && !trimmed.ends_with('{') && !ctx.LineHasComment(i + 1) && !ctx.LineHasMultiLineToken(i + 1); + && !trimmed.ends_with('{') && !ctx.Line(i + 1).contains("//") && !ctx.Line(i + 1).contains("R\""); if (candidate) { std::string joined(ctx.Line(i + 1)); std::size_t j = i + 1; @@ -739,7 +736,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { // lambda body starts — leave those wrapped. bool closes = delta + ParenDelta(stripped[j]) <= 0; if (next.empty() || (next.ends_with('{') && !closes) - || ctx.LineHasComment(j + 1) || ctx.LineHasMultiLineToken(j + 1)) { + || ctx.Line(j + 1).contains("//") || ctx.Line(j + 1).contains("R\"")) { ok = false; break; } @@ -760,7 +757,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { continue; } } else if (delta == 0 && i + 1 < ctx.lines.size() && !trimmed.starts_with('#') - && !ctx.LineHasComment(i + 1) && !ctx.LineHasMultiLineToken(i + 1)) { + && !ctx.Line(i + 1).contains("//") && !ctx.Line(i + 1).contains("R\"")) { auto isOpStart = [](std::string_view s) { return s.starts_with("&&") || s.starts_with("||") || s.starts_with("| "); }; @@ -774,7 +771,7 @@ inline void AddProjectLintRules(Crafter::Configuration& cfg) { std::size_t j = i + 1; bool ok = true; while (j < ctx.lines.size() && isOpStart(Trim(stripped[j]))) { - if (ParenDelta(stripped[j]) != 0 || ctx.LineHasComment(j + 1) || ctx.LineHasMultiLineToken(j + 1)) { + if (ParenDelta(stripped[j]) != 0 || ctx.Line(j + 1).contains("//") || ctx.Line(j + 1).contains("R\"")) { ok = false; break; } diff --git a/tests/HouseRules/main.cpp b/tests/HouseRules/main.cpp index 7afa0a5..0b3593e 100644 --- a/tests/HouseRules/main.cpp +++ b/tests/HouseRules/main.cpp @@ -252,49 +252,6 @@ int main() { Check(second.summary.changedFiles.empty(), "wrap-join is idempotent after the fixpoint"); } - // The guards used to be substring probes over the raw line, so a literal - // whose TEXT contained `//` or `R"` looked like a comment or a raw string - // and silently disabled the rule. Both shapes occur in this repo's own - // sources — "MARKER" ends in the characters R", and any string mentioning - // a lint-disable directive contains //. - { - RuleRun r = RunRule("void F() {\n" " auto hit = text.find(\"MARKER\",\n" " start);\n" "}\n", "wrap-join", LintMode::Apply); - Check(r.text.contains("text.find(\"MARKER\", start);"), "R\" inside a literal no longer blocks wrap-join"); - } - { - RuleRun r = RunRule("void F() {\n" " Check(read() == \"// lint-disable-next-line trim\\n\",\n" " \"message\");\n" "}\n", "wrap-join", LintMode::Apply); - Check(r.text.contains("\\n\", \"message\");"), "// inside a literal no longer blocks wrap-join"); - } - - // A real trailing comment still blocks the join: text pulled up past a - // `//` would be swallowed by it. - { - RuleRun r = RunRule("void F() {\n" " auto v = g(alpha, // why\n" " beta);\n" "}\n", "wrap-join", LintMode::Apply); - Check(r.text.contains("g(alpha, // why\n"), "a real comment still blocks wrap-join"); - } - - // A genuinely multi-line literal is never reflowed — joining its lines - // would change the string's contents. - { - std::string_view source = "void F() {\n" - " auto text = R\"sql(SELECT a,\n" - " b FROM t)sql\";\n" - "}\n"; - RuleRun r = RunRule(source, "wrap-join", LintMode::Apply); - Check(r.text == source, "wrap-join leaves a multi-line raw string alone"); - RuleRun paren = RunRule(source, "paren-spacing", LintMode::Apply); - Check(paren.text == source, "paren-spacing leaves a multi-line raw string alone"); - } - - // Type keywords inside a literal are text, not declarations. - { - std::string_view source = "void F() {\n" - " auto sql = R\"q(int x; unsigned long y;)q\";\n" - "}\n"; - RuleRun r = RunRule(source, "fixed-width-types", LintMode::Apply); - Check(r.text == source, "fixed-width-types leaves type names inside a raw string alone"); - } - if (Failures > 0) { std::println(std::cerr, "{} assertions failed", Failures); return 1; diff --git a/tests/Lint/main.cpp b/tests/Lint/main.cpp index fa81ec4..c900b27 100644 --- a/tests/Lint/main.cpp +++ b/tests/Lint/main.cpp @@ -156,7 +156,8 @@ int main() { Configuration cfg = FixtureConfig(); cfg.AddLintRule("marker", [](LintContext& ctx) { const std::string& code = ctx.CommentStripped(); - for (std::size_t pos = code.find("MARKER"); pos != std::string::npos; pos = code.find("MARKER", pos + 1)) { + for (std::size_t pos = code.find("MARKER"); pos != std::string::npos; + pos = code.find("MARKER", pos + 1)) { std::size_t line = 1 + std::count(code.begin(), code.begin() + pos, '\n'); ctx.Report(line, "MARKER in code"); } @@ -369,7 +370,8 @@ int main() { Configuration cfg = s.Config({"a"}); AddTrimRule(cfg); RunLint(cfg, Mode(LintMode::Apply)); - Check(s.Read("a") == "// lint-disable-next-line trim\nkeep \ntrim\n", "suppressed line keeps its bytes; the unsuppressed one is fixed"); + Check(s.Read("a") == "// lint-disable-next-line trim\nkeep \ntrim\n", + "suppressed line keeps its bytes; the unsuppressed one is fixed"); } // Multiple rule names on one directive (space- or comma-separated). @@ -425,136 +427,6 @@ int main() { Check(s.Read("a") == "// lint-disable-file flag\nbad\n", "other rules still format"); } - // ---------------- token layer ---------------- - // - // The source is spelled with escaped literals rather than a raw string so - // that this file stays lintable by the very rules under test; the scratch - // file it writes does contain a genuine multi-line raw string. - { - constexpr std::string_view Source = - "#ifdef CRAFTER_LINT_NEVER_DEFINED\n" // 1 - "void HiddenBranch();\n" // 2 - "#endif\n" // 3 - "// a real comment\n" // 4 - "int url = 1; // https://example.com\n" // 5 - "auto raw = R\"raw(spans lines\n" // 6 - " // not a comment\n" // 7 - " int notADecl;\n" // 8 - ")raw\";\n"; // 9 - - Scratch s("tokens"); - s.Write("f", Source); - Configuration cfg = s.Config({"f"}); - cfg.AddLintRule("tokens", [](LintContext& ctx) { - std::span toks = ctx.Tokens(); - Check(!toks.empty(), "tokens: file lexes to a non-empty stream"); - - // Every token's offset/length must address its own bytes, or a - // transform editing at an offset would corrupt the file. - bool offsetsSound = true; - for (const LintToken& t : toks) { - if (t.offset + t.length > ctx.content.size() || ctx.TokenText(t).empty()) offsetsSound = false; - } - Check(offsetsSound, "tokens: every offset/length addresses real bytes"); - - // Ordered by offset, so binary search in TokensOnLine is valid. - bool ordered = std::ranges::is_sorted(toks, {}, &LintToken::offset); - Check(ordered, "tokens: stream is in source order"); - - // Inactive #ifdef branch is still lexed — this is what keeps token - // rules covering every platform, unlike an AST. - bool sawHidden = std::ranges::any_of(toks, [&](const LintToken& t) { - return t.kind == LintTokenKind::Identifier && ctx.TokenText(t) == "HiddenBranch"; - }); - Check(sawHidden, "tokens: inactive #ifdef branch is lexed"); - - // A multi-line raw string is exactly one Literal, comment markers - // and declarations inside it included. - auto isRaw = [&](const LintToken& t) { return ctx.TokenText(t).starts_with("R\"raw("); }; - Check(std::ranges::count_if(toks, isRaw) == 1, "tokens: raw string is a single token"); - auto raw = std::ranges::find_if(toks, isRaw); - if (raw != toks.end()) { - Check(raw->kind == LintTokenKind::Literal, "tokens: raw string is a Literal"); - Check(raw->line == 6, "tokens: raw string starts on line 6"); - Check(ctx.TokenText(*raw).contains("// not a comment"), "tokens: raw string body kept intact"); - Check(ctx.TokenText(*raw).ends_with(")raw\""), "tokens: raw string spans to its own terminator"); - } - - // The only comments are the two real ones on lines 4 and 5 — the - // `//` on line 7 lives inside the raw string. - std::vector commentLines; - for (const LintToken& t : toks) { - if (t.kind == LintTokenKind::Comment) commentLines.push_back(t.line); - } - Check(commentLines == std::vector{4, 5}, "tokens: only real comments are Comment tokens"); - Check(ctx.LineHasComment(4), "tokens: LineHasComment finds a whole-line comment"); - Check(ctx.LineHasComment(5), "tokens: LineHasComment finds a trailing comment"); - Check(!ctx.LineHasComment(7), "tokens: `//` inside a raw string is not a comment"); - Check(!ctx.LineHasComment(2), "tokens: code-only line has no comment"); - - // TokensOnLine brackets by starting line. - std::span line2 = ctx.TokensOnLine(2); - Check(!line2.empty() && ctx.TokenText(line2.front()) == "void", "tokens: TokensOnLine starts at the line's first token"); - Check(std::ranges::all_of(line2, [](const LintToken& t) { return t.line == 2; }), "tokens: TokensOnLine stays on its line"); - - // SetContent must invalidate the cache, or offsets point into a - // buffer that no longer exists. - ctx.SetContent("int replaced;\n"); - std::span after = ctx.Tokens(); - Check(!after.empty() && ctx.TokenText(after.front()) == "int", "tokens: re-lexed after SetContent"); - Check(std::ranges::none_of(after, [&](const LintToken& t) { return ctx.TokenText(t) == "HiddenBranch"; }), "tokens: stale tokens are dropped after SetContent"); - }); - RunLint(cfg, Mode(LintMode::Report)); - } - - // CommentStripped over a raw string holding an ODD number of quotes. The - // character-scanning version treated R"( as an ordinary string open, so the - // quote inside the body closed it early and every following line was - // swallowed as literal text — code after the raw string vanished from the - // stripped view. Lexing gets the extent right. - { - constexpr std::string_view Source = - "auto banner = R\"(he said \"hi)\";\n" // 1: one quote inside the body - "int afterRaw = 2;\n" // 2: must survive as code - "// MARKER comment\n" // 3 - "auto plain = \"MARKER text\";\n"; // 4 - - Scratch s("strip-rawstring"); - s.Write("f", Source); - Configuration cfg = s.Config({"f"}); - cfg.AddLintRule("strip", [](LintContext& ctx) { - const std::string& code = ctx.CommentStripped(); - Check(code.size() == ctx.content.size(), "strip: byte length preserved"); - Check(std::ranges::count(code, '\n') == std::ranges::count(ctx.content, '\n'), "strip: newlines preserved"); - Check(code.contains("afterRaw"), "strip: code after an odd-quoted raw string survives"); - Check(!code.contains("he said"), "strip: raw string body is blanked"); - Check(!code.contains("MARKER"), "strip: comment and literal bodies are blanked"); - Check(code.contains("auto plain ="), "strip: code around a literal survives"); - // The raw string collapses to R"…" — exactly two quotes, so rules - // that bracket a literal by counting quotes still work. - std::string_view line1 = std::string_view(code).substr(0, code.find('\n')); - Check(std::ranges::count(line1, '"') == 2, "strip: raw string leaves exactly two quotes"); - }); - RunLint(cfg, Mode(LintMode::Report)); - } - - // Non-C++ extensions are not lexed: GLSL through a C++ lexer would produce - // plausible-looking nonsense rather than an honest refusal. - { - Scratch s("tokens-foreign"); - fs::path shader = s.dir / "f.frag"; - { - std::ofstream f(shader, std::ios::binary | std::ios::trunc); - f << "#version 450\nvoid main() { }\n"; - } - Configuration cfg = s.Config({}); - cfg.shaders.emplace_back(fs::path(shader), "main", ShaderType::Fragment); - cfg.AddLintRule("no-lex", [](LintContext& ctx) { - Check(ctx.Tokens().empty(), "tokens: shaders are not lexed as C++"); - }); - RunLint(cfg, Mode(LintMode::Report)); - } - if (Failures > 0) { std::println(std::cerr, "{} assertions failed", Failures); return 1;