/*! \file ast.cpp * \brief AST-based expression parser and evaluator. * * See ast.h for design constraints and public API. */ #include "copyright.h" #include "autoconf.h" #include "config.h" #include "externs.h" #include "ast.h" #include "functions.h" #include #include #include #include #include #include "jit_tier1_stamp.h" // Tier 1 build stamp for this unit (#2061). Folded into the persisted // code_cache's staleness key so a codegen change here invalidates entries // compiled by the previous build. Updates when THIS unit is recompiled, // which is what makes it work under incremental make. TIER1_STAMP_DEFINE(TIER1_STAMP_AST); // --------------------------------------------------------------- // Tokenizer — implemented in ast_scan.cpp (generated by Ragel // from ast_scan.rl). // --------------------------------------------------------------- // --------------------------------------------------------------- // Parser // --------------------------------------------------------------- static std::string ast_upper_builtin_name(std::string_view funcName) { LBuf TempFun = LBuf_Src("ast_upper_builtin"); size_t nName = funcName.size(); if (nName >= LBUF_SIZE) { nName = LBUF_SIZE - 1; } memcpy(TempFun, funcName.data(), nName); TempFun[nName] = '\0'; size_t nUpper; UTF8 *pUpper = mux_strupr(TempFun, nUpper); if (nUpper >= LBUF_SIZE) { nUpper = LBUF_SIZE - 1; } return std::string(reinterpret_cast(pUpper), nUpper); } static ASTNoevalKind ast_noeval_kind(std::string_view funcName) { const std::string name = ast_upper_builtin_name(funcName); if (name == "IF" || name == "IFELSE") { return ASTNOEVAL_IFELSE; } if (name == "ITER") { return ASTNOEVAL_ITER; } if (name == "CAND" || name == "CANDBOOL") { return (name == "CANDBOOL") ? ASTNOEVAL_CANDBOOL : ASTNOEVAL_CAND; } if (name == "COR" || name == "CORBOOL") { return (name == "CORBOOL") ? ASTNOEVAL_CORBOOL : ASTNOEVAL_COR; } if (name == "SWITCH") { return ASTNOEVAL_SWITCH; } if (name == "CASE") { return ASTNOEVAL_CASE; } if (name == "SWITCHALL") { return ASTNOEVAL_SWITCHALL; } if (name == "CASEALL") { return ASTNOEVAL_CASEALL; } if (name == "ULAMBDA") { return ASTNOEVAL_ULAMBDA; } return ASTNOEVAL_NONE; } static bool ast_noeval_arg_is_deferred(ASTNoevalKind kind, int argIndex, int nfargs) { switch (kind) { case ASTNOEVAL_IFELSE: return argIndex >= 1; case ASTNOEVAL_ITER: return argIndex == 1; case ASTNOEVAL_CAND: case ASTNOEVAL_CANDBOOL: case ASTNOEVAL_COR: case ASTNOEVAL_CORBOOL: return true; case ASTNOEVAL_SWITCH: case ASTNOEVAL_CASE: case ASTNOEVAL_SWITCHALL: case ASTNOEVAL_CASEALL: if (argIndex == 0) { return false; } if ((nfargs % 2) == 0 && argIndex == nfargs - 1) { return true; } return (argIndex % 2) == 0; case ASTNOEVAL_ULAMBDA: return argIndex == 0; // only the body/reference is deferred case ASTNOEVAL_NONE: default: return false; } } static bool ast_call_arg_is_deferred(const ASTNode *call, int argIndex) { if (!call || argIndex < 0) { return false; } if (argIndex < static_cast(call->deferred_args.size())) { return call->deferred_args[argIndex].is_deferred; } if (!call->parser_known_noeval) { return false; } const int nfargs = static_cast(call->children.size()); return ast_noeval_arg_is_deferred(call->noeval_kind, argIndex, nfargs); } static const std::string *ast_call_raw_arg(const ASTNode *call, int argIndex) { if (!call || argIndex < 0) { return nullptr; } if (argIndex < static_cast(call->deferred_args.size())) { return &call->deferred_args[argIndex].raw_text; } return nullptr; } // Hard cap on AST *parser* recursion depth, mirroring the evaluator's // AST_EVAL_MAX_DEPTH (below). parseSequence/parseEvalBracket/parseBraceGroup/ // parseFunctionCall are mutually recursive on the nesting of [], {}, and () in // the (LBUF-bounded) input. On a default 8 MiB stack the LBUF cap already keeps // even maximally nested input ("[[[[...]]]]") from overflowing, but — like the // evaluator's cap — this bounds adversarial deep nesting independently of stack // size (the evaluator's comment notes platforms with smaller stack defaults). // The counter is thread_local so it also bounds the nested re-parse of NOEVAL // structural arguments (ast_parse_region in parser_apply_structural_arg_policy). // // Over the cap, parseSequence returns its (empty) node without recursing; the // enclosing parseEvalBracket/etc. still consumed their opening token, so m_pos // always advances and parsing terminates. Real softcode nests a few levels; // 1000 is far beyond legitimate use and well under the evaluator's reach. // static constexpr int AST_PARSE_MAX_DEPTH = 1000; static thread_local int s_ast_parse_depth = 0; namespace { class AstParseDepthGuard { public: AstParseDepthGuard() { ++s_ast_parse_depth; } ~AstParseDepthGuard() { --s_ast_parse_depth; } bool overflow() const { return s_ast_parse_depth > AST_PARSE_MAX_DEPTH; } }; } class ASTParser { public: ASTParser(const std::vector &tokens) : m_tokens(tokens), m_pos(0), m_bracketDepth(0), m_braceDepth(0) {} std::unique_ptr parse() { return parseSequence(false, false, false, false); } private: const std::vector &m_tokens; size_t m_pos; int m_bracketDepth; int m_braceDepth; const ASTToken &peek() const { return m_tokens[m_pos]; } ASTToken advance() { return m_tokens[m_pos++]; } bool atEnd() const { return m_pos >= m_tokens.size() || m_tokens[m_pos].type == ASTTOK_EOF; } static bool parser_lookup_builtin_noeval(std::string_view funcName) { LBuf TempFun = LBuf_Src("lookup_noeval"); size_t nName = funcName.size(); if (nName >= LBUF_SIZE) { nName = LBUF_SIZE - 1; } memcpy(TempFun, funcName.data(), nName); TempFun[nName] = '\0'; size_t nUpper; UTF8 *pUpper = mux_strupr(TempFun, nUpper); if (nUpper >= LBUF_SIZE) { nUpper = LBUF_SIZE - 1; } memcpy(TempFun, pUpper, nUpper); TempFun[nUpper] = '\0'; std::vector name_key(TempFun.get(), TempFun.get() + nUpper); const auto it = mudstate.builtin_functions.find(name_key); return it != mudstate.builtin_functions.end() && (it->second->flags & FN_NOEVAL) != 0; } static bool parser_should_structuralize_arg(const ASTNode *call, int argIndex) { return ast_call_arg_is_deferred(call, argIndex); } void parser_capture_raw_arg(ASTNode *call, size_t start, size_t end) { if (!call || !call->parser_known_noeval) { return; } call->deferred_args.emplace_back(rawTextFromTokens(start, end), false); } void parser_apply_structural_arg_policy(ASTNode *call) { if (!call || !call->parser_known_noeval) { return; } call->deferred_args.resize(call->children.size()); for (int i = 0; i < static_cast(call->children.size()); i++) { if (!parser_should_structuralize_arg(call, i)) { continue; } call->deferred_args[i].is_deferred = true; const std::string *raw = ast_call_raw_arg(call, i); if (!raw) { continue; } auto structural = ast_parse_region( ASTSourceSpan(reinterpret_cast(raw->c_str()), raw->size()), ASTLEX_STRUCTURAL); if (structural) { call->children[i] = std::move(structural); } } } std::string rawTextFromTokens(size_t start, size_t end) const { std::string raw; for (size_t i = start; i < end && i < m_tokens.size(); i++) { raw.append(m_tokens[i].text.data(), m_tokens[i].text.size()); } return raw; } std::unique_ptr parseSequence( bool stopRP, bool stopRB, bool stopRC, bool stopCM) { // Bound parser recursion: every []/{}/() nesting level re-enters here. // AstParseDepthGuard depth_guard; auto seq = std::make_unique(AST_SEQUENCE); if (depth_guard.overflow()) { return seq; } // Depth of bare parentheses opened inside this sequence (#1219). // // A '(' that follows a name is consumed by parseFuncCall, which // recurses and accounts for its own parens; the only thing counted // here is a parenthesis appearing as literal text. While one is // open, a ')' or ',' belongs to it rather than to the enclosing // call, so neither terminates the sequence. // // 2.13 did this with a stack of expected closers in parse_to_lite: // '(' pushed ')', and a closer found on the stack unwound to it // rather than ending the argument. Without the equivalent here, // `strcat(Meet me (Tue, 5pm) downtown)` ended its argument at the // first ')' and the comma split it, giving // `Meet me (Tue5pm downtown)`. // int parenDepth = 0; while (!atEnd()) { ASTTokenType t = peek().type; if (stopRP && t == ASTTOK_RPAREN && 0 == parenDepth) break; if (stopRB && t == ASTTOK_RBRACK) break; if (stopRC && t == ASTTOK_RBRACE) break; if (stopCM && t == ASTTOK_COMMA && 0 == parenDepth) break; if (ASTTOK_LPAREN == t) { parenDepth++; } else if (ASTTOK_RPAREN == t && 0 < parenDepth) { parenDepth--; } auto node = parseOne(); if (node) { seq->addChild(std::move(node)); } } if (seq->children.size() == 1) { return std::move(seq->children[0]); } return seq; } std::unique_ptr parseOne() { const ASTToken &tok = peek(); switch (tok.type) { case ASTTOK_LIT: { auto n = std::make_unique(AST_LITERAL, tok.text); advance(); return n; } case ASTTOK_SPACE: { auto n = std::make_unique(AST_SPACE, tok.text); advance(); return n; } case ASTTOK_PCT: { auto n = std::make_unique(AST_SUBST, tok.text); advance(); // No DynCall support — if ( follows a substitution, // the ( is just a literal parenthesis. return n; } case ASTTOK_ESC: { auto n = std::make_unique(AST_ESCAPE, tok.text); advance(); return n; } case ASTTOK_SEMI: { auto n = std::make_unique(AST_SEMICOLON, tok.text); advance(); return n; } case ASTTOK_FUNC: return parseFuncCall(); case ASTTOK_LBRACK: return parseEvalBracket(); case ASTTOK_LBRACE: return parseBraceGroup(); case ASTTOK_RPAREN: case ASTTOK_RBRACK: case ASTTOK_RBRACE: case ASTTOK_COMMA: case ASTTOK_LPAREN: { auto n = std::make_unique(AST_LITERAL, tok.text); advance(); return n; } case ASTTOK_EOF: return nullptr; } return nullptr; } std::unique_ptr parseFuncCall() { ASTToken funcTok = advance(); auto call = std::make_unique(AST_FUNCCALL, funcTok.text); call->noeval_kind = ast_noeval_kind(funcTok.text); call->parser_known_noeval = parser_lookup_builtin_noeval(funcTok.text); if (atEnd() || peek().type != ASTTOK_LPAREN) { call->type = AST_LITERAL; return call; } advance(); // consume LPAREN parseArgList(call.get()); return call; } void parseArgList(ASTNode *call) { // Handle zero-argument function calls: foo() // if (!atEnd() && peek().type == ASTTOK_RPAREN) { advance(); call->has_close_paren = true; return; } call->has_close_paren = false; // When inside an eval bracket, ] terminates the argument // list (matching mux_exec behavior where ] closes the bracket // even with unclosed parentheses). Outside brackets, ] is // literal text in function arguments. // bool inBracket = (m_bracketDepth > 0); bool inBrace = (m_braceDepth > 0); size_t argStart = m_pos; auto arg = parseSequence(true, inBracket, inBrace, true); parser_capture_raw_arg(call, argStart, m_pos); call->addChild(std::move(arg)); while (!atEnd() && peek().type == ASTTOK_COMMA) { advance(); argStart = m_pos; arg = parseSequence(true, inBracket, inBrace, true); parser_capture_raw_arg(call, argStart, m_pos); call->addChild(std::move(arg)); } if (!atEnd() && peek().type == ASTTOK_RPAREN) { advance(); call->has_close_paren = true; } parser_apply_structural_arg_policy(call); } std::unique_ptr parseEvalBracket() { advance(); // consume LBRACK m_bracketDepth++; auto bracket = std::make_unique(AST_EVALBRACKET); auto contents = parseSequence(false, true, false, false); bracket->addChild(std::move(contents)); if (!atEnd() && peek().type == ASTTOK_RBRACK) { advance(); } else { bracket->has_close_bracket = false; } m_bracketDepth--; return bracket; } std::unique_ptr parseBraceGroup() { advance(); // consume LBRACE m_braceDepth++; auto group = std::make_unique(AST_BRACEGROUP); auto contents = parseSequence(false, false, true, false); group->addChild(std::move(contents)); if (!atEnd() && peek().type == ASTTOK_RBRACE) { advance(); } else { group->has_close_brace = false; } m_braceDepth--; return group; } }; // --------------------------------------------------------------- // Public parse API // --------------------------------------------------------------- std::unique_ptr ast_parse(const std::vector &tokens) { ASTParser parser(tokens); return parser.parse(); } std::unique_ptr ast_parse_region(ASTSourceSpan span, ASTLexMode mode) { auto tokens = ast_tokenize_mode(span.input, span.nLen, mode); return ast_parse(tokens); } std::unique_ptr ast_parse_string(const UTF8 *input, size_t nLen) { return ast_parse_region(ASTSourceSpan(input, nLen), ASTLEX_EVAL); } // --------------------------------------------------------------- // Utility functions // --------------------------------------------------------------- std::string ast_raw_text(const ASTNode *n) { if (!n) { return ""; } switch (n->type) { case AST_LITERAL: case AST_SPACE: case AST_SUBST: case AST_ESCAPE: return n->text; case AST_SEMICOLON: return ";"; case AST_FUNCCALL: { std::string r = n->text + "("; for (size_t i = 0; i < n->children.size(); i++) { if (i > 0) r += ","; r += ast_raw_text(n->children[i].get()); } if (n->has_close_paren) { r += ")"; } return r; } case AST_EVALBRACKET: { std::string r = "["; for (const auto &c : n->children) { r += ast_raw_text(c.get()); } return r + "]"; } case AST_BRACEGROUP: { std::string r = "{"; for (const auto &c : n->children) { r += ast_raw_text(c.get()); } return r + "}"; } case AST_SEQUENCE: { std::string r; for (const auto &c : n->children) { r += ast_raw_text(c.get()); } return r; } } return ""; } static const char *ast_node_name(ASTNodeType t) { switch (t) { case AST_SEQUENCE: return "Seq"; case AST_LITERAL: return "Lit"; case AST_SPACE: return "Sp"; case AST_SUBST: return "Sub"; case AST_ESCAPE: return "Esc"; case AST_FUNCCALL: return "Call"; case AST_EVALBRACKET: return "Eval"; case AST_BRACEGROUP: return "Brace"; case AST_SEMICOLON: return "Semi"; } return "???"; } void ast_dump(const ASTNode *node, int indent) { if (!node) { return; } // Bound dump recursion so debug logging on a pathological AST // can't overflow the stack. Use a fixed cap rather than the // evaluator's depth counter since dump is diagnostic. // // Indent by hand rather than with "%*s". mux_vsnprintf implements the // '-' and '0' flags and literal digit widths only -- it has no '*' width, // so "%*s" reached mux_assert(0) and abort()ed the process (#1429 covers // the general case; this was the one live instance in the tree). Nothing // calls ast_dump today, which is the only reason it never fired, but it // is declared in ast.h as a debug helper and is meant to be callable. // UTF8 pad[81]; size_t nPad = (indent < 0) ? 0 : static_cast(indent); if (sizeof(pad) - 1 < nPad) { nPad = sizeof(pad) - 1; } memset(pad, ' ', nPad); pad[nPad] = '\0'; if (indent > 2 * 400) { STARTLOG(LOG_DEBUG, "AST", "DUMP"); Log.tinyprintf(T("%s... (truncated)"), pad); ENDLOG; return; } STARTLOG(LOG_DEBUG, "AST", "DUMP"); Log.tinyprintf(T("%s%s"), pad, ast_node_name(node->type)); if (!node->text.empty()) { Log.tinyprintf(T(" \"%s\""), node->text.c_str()); } ENDLOG; for (const auto &child : node->children) { ast_dump(child.get(), indent + 2); } } // --------------------------------------------------------------- // AST Evaluator (Phase 2) // --------------------------------------------------------------- // // Walks the AST tree and produces evaluated output into buff/bufc. // Uses the same runtime context as mux_exec (mudstate, mudconf, // builtin function table, registers, iterator stack, etc.) // // For %-substitutions, delegates to mux_exec on the short // substitution text (2-6 bytes). This gives exact compatibility // with all L2 dispatch table entries without reimplementing them. // // For function calls, looks up in mudstate.builtin_functions and // dispatches through the existing FUN handler. NOEVAL functions // receive raw text via ast_raw_text() — the handler calls mux_exec // internally as before. // // Replicate 2.13's noeval pass on an AST subtree. // // In 2.13, mux_exec with EV_EVAL off still processes backslash // escapes (the handler has no EV_EVAL guard). Percent substitutions // are copied literally (guarded by EV_EVAL). // // This produces a string with one layer of backslash stripping, // which can then be re-tokenized and evaluated. // // Forward declaration. // static void ast_eval_node(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs); // Evaluate a selected argument from a FN_NOEVAL function using the // 2.13-style noeval pass followed by reparse/re-eval. // // This replicates the two-pass behavior observed in 2.13: // Pass 1: noeval -- parse/evaluate the deferred region under ASTLEX_NOEVAL // Pass 2: eval -- re-tokenize the result and evaluate it // static void ast_eval_deferred_region(const ASTNode *node, const ASTNode *noevalNode, const std::string *rawText, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if (!node) { return; } std::string text; if (noevalNode) { LBuf temp = LBuf_Src("ast_noeval_region"); UTF8 *tp = temp; ast_eval_node(noevalNode, temp, &tp, executor, caller, enactor, ((eval & ~(EV_EVAL | EV_TOP | EV_FMAND | EV_STRIP_CURLY | EV_FCHECK)) | EV_NOFCHECK), cargs, ncargs); *tp = '\0'; text.assign(reinterpret_cast(temp.get()), tp - temp); } else if (rawText) { auto noevalAst = ast_parse_region( ASTSourceSpan(reinterpret_cast(rawText->c_str()), rawText->size()), ASTLEX_NOEVAL); if (!noevalAst) { return; } LBuf temp = LBuf_Src("ast_noeval_region"); UTF8 *tp = temp; ast_eval_node(noevalAst.get(), temp, &tp, executor, caller, enactor, ((eval & ~(EV_EVAL | EV_TOP | EV_FMAND | EV_STRIP_CURLY | EV_FCHECK)) | EV_NOFCHECK), cargs, ncargs); *tp = '\0'; text.assign(reinterpret_cast(temp.get()), tp - temp); } else { return; } // The selected branch/body of a FN_NOEVAL function is evaluated // with one outer brace layer stripped. Nested inner braces remain // part of the deferred text. if ( node->type == AST_BRACEGROUP && text.size() >= 2 && text.front() == '{' && text.back() == '}') { text = text.substr(1, text.size() - 2); } // Pass 2: re-tokenize the selected region in EVAL mode and // evaluate the resulting subtree directly. This is the first // concrete use of the parser-controlled region-parse API. // auto reparsed = ast_parse_region( ASTSourceSpan(reinterpret_cast(text.c_str()), text.size()), ASTLEX_EVAL); if (!reparsed) { return; } ast_eval_node(reparsed.get(), buff, bufc, executor, caller, enactor, (eval & ~(EV_TOP | EV_FMAND | EV_STRIP_CURLY)) | EV_EVAL | EV_FCHECK, cargs, ncargs); } static bool ast_is_malformed_qsubst(const ASTNode *node) { if ( !node || node->type != AST_SUBST) { return false; } const std::string &txt = node->text; return txt.size() >= 3 && txt[0] == '%' && (txt[1] == 'q' || txt[1] == 'Q') && txt[2] == '<' && txt.find('>', 3) == std::string::npos; } // Evaluate a function argument with the same top-level space trimming // that parse_arglist()/parse_to() applies around comma-separated args. // Evaluate children [first, last) of a SEQUENCE, applying 2.13's // one-call-per-region rule. // // EV_FCHECK without EV_FMAND means "the first '(' in this region may be // a function call". 2.13 clears EV_FCHECK once that opportunity is used // (mux/src/eval.cpp:1677, `eval &= ~EV_FCHECK`), so a later call in the // same region emits as literal text: // // [strcat(x add(1,2) y)] -> x add(1,2) y (not "x 3 y") // // Two places evaluate such a run — the AST_SEQUENCE case in // ast_eval_node and the space-compressed argument path in // ast_eval_argument — and both have to apply the rule. Only the former // did, so the rule was dead for exactly the case that matters, a call // sitting mid-argument (#1214). Shared here so the two cannot drift. // // Set when a lookup fails under EV_FMAND, to stop the rest of the region // from being emitted (#1247). 2.13 does this with `*bufc = oldp; break;` // in its single eval loop (mux/src/eval.cpp:1497); 2.14 evaluates a tree, // so the failure has to be signalled from the funccall node up to the // sequence that contains it. // // Scoped the same way as the FCHECK opportunity: saved and cleared when // entering an eval-bracket region, consumed by the nearest enclosing // EV_FMAND sequence, restored on the way out. That keeps an inner // region's failure from aborting an outer one, and keeps a failure inside // a u()'d attribute from escaping into its caller. // static bool s_fmand_abort = false; static void ast_eval_sequence_children(const ASTNode *node, size_t first, size_t last, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { // Whether a call in this region can still be recognised. // bool armed = (eval & EV_FCHECK) != 0; // Any non-space child spends it. 2.13 checks only the first '(' // it reaches, and a dispatched call clears EV_FCHECK for whatever // follows (mux/src/eval.cpp:1677): // // [strcat(x add(1,2) y)] -> x add(1,2) y // [add(1,2) mul(3,4)] -> 3 mul(3,4) // [add(1,2) zz mul(3,4)] -> 3 zz mul(3,4) // // The rule is uniform across region kinds. #1238 briefly made // EV_FMAND regions special so that text before a call would not // spend the opportunity: 2.13's candidate name there is the // accumulated OUTPUT, and it folds preceding text into the name // ([x add(1,2) y] -> "#-1 FUNCTION (X ADD) NOT FOUND"). That // existed to avoid settling an unadjudicated shape by accident. // // It has since been settled the other way (#1246): the span-name // error is not the target, literal text is the better answer, and // the compiled route — which never modelled the span — is the // reference. With the distinction gone both routes agree: // // [x add(1,2) y] -> x add(1,2) y // [zz mul(2,3)] -> zz mul(2,3) // for (size_t i = first; i < last; i++) { const ASTNode *child = node->children[i].get(); int childEval = eval; if ( !armed || child->type != AST_FUNCCALL) { childEval = eval & ~EV_FCHECK; } ast_eval_node(child, buff, bufc, executor, caller, enactor, childEval, cargs, ncargs); // A failed mandatory lookup ends the region. Everything after it // would be literal text by both engines' rules anyway (a dispatched // call already spends the recognition opportunity), so nothing // evaluable is lost -- but emitting it makes the diagnostic read as // though part of the region had succeeded (#1247). // if ( (eval & EV_FMAND) && s_fmand_abort) { s_fmand_abort = false; break; } if ( armed && child->type != AST_SPACE) { armed = false; } } } static void ast_eval_argument(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if (!node) { return; } if ( node->type == AST_SEQUENCE && !node->children.empty() && mudconf.space_compress && !(eval & EV_NO_COMPRESS)) { size_t first = 0; size_t last = node->children.size(); while ( first < last && node->children[first]->type == AST_SPACE) { first++; } while ( last > first && node->children[last - 1]->type == AST_SPACE) { last--; } ast_eval_sequence_children(node, first, last, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return; } ast_eval_node(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } // --------------------------------------------------------------- // Native %-substitution handler // --------------------------------------------------------------- // // Handles all L2 dispatch table substitutions natively instead of // delegating to mux_exec. The node->text is the full %-sequence // as gathered by gather_pct (e.g. "%0", "%qa", "%q", "%xn", // "%c", "%va", "%i0", "%=", "%%", "%r", "%b", etc.) // static void ast_eval_subst(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { const std::string &txt = node->text; if (txt.size() < 2) { // Bare '%' at end of string — output literally. // safe_chr('%', buff, bufc); return; } // ## / #@ / #$ — iterator/switch substitutions. // if (txt[0] == '#') { switch (txt[1]) { case '#': // ## — bound variable (same as %i0). // { int i = mudstate.in_loop - 1; if (0 <= i && i < MAX_ITEXT && mudstate.itext[i]) { safe_str(mudstate.itext[i], buff, bufc); } } break; case '@': // #@ — list place number (same as inum()). // { int i = mudstate.in_loop - 1; if (0 <= i && i < MAX_ITEXT) { safe_ltoa(mudstate.inum[i], buff, bufc); } } break; case '$': // #$ — switch value (same as switch() matched value). // Resolved from mudstate.switch_token if available. // if (mudstate.switch_token) { safe_str(mudstate.switch_token, buff, bufc); } break; } return; } unsigned char ch = static_cast(txt[1]); unsigned char upper = static_cast(mux_toupper_ascii(ch)); // The L2 table sets flag 0x80 on uppercase A, M, N, O, P, Q, S, V // to trigger mux_toupper_first on the substituted value. // bool bUpperCase = false; if ('A' <= ch && ch <= 'Z') { switch (ch) { case 'A': case 'M': case 'N': case 'O': case 'P': case 'Q': case 'S': case 'V': bUpperCase = true; break; } } LBuf scratch = LBuf_Src("ast_eval_subst"); UTF8 *TempPtr = *bufc; // %0-%9 — command argument substitution. // if (ch >= '0' && ch <= '9') { int i = ch - '0'; if (i < ncargs && cargs[i]) { safe_str(cargs[i], buff, bufc); } return; } switch (upper) { case 'Q': // %q0-%q9, %qa-%qz, %q — register substitution. // if (txt.size() >= 3) { if (txt[2] == '<') { // Named register: %q // size_t close = txt.find('>', 3); if (close != std::string::npos) { size_t nName = close - 3; const UTF8 *pName = reinterpret_cast(txt.c_str() + 3); int regnum = -1; if ( 1 == nName && (regnum = mux_RegisterSet[pName[0]]) >= 0 && regnum < MAX_GLOBAL_REGS) { if ( mudstate.global_regs[regnum] && mudstate.global_regs[regnum]->reg_len > 0) { safe_copy_buf(mudstate.global_regs[regnum]->reg_ptr, mudstate.global_regs[regnum]->reg_len, buff, bufc); } } else if (IsValidNamedReg(pName, nName)) { reg_ref *rr = NamedRegRead(mudstate.named_regs, pName, nName); if (rr && rr->reg_len > 0) { safe_copy_buf(rr->reg_ptr, rr->reg_len, buff, bufc); } } } else { // Malformed %q — output literally. // safe_str(reinterpret_cast(txt.c_str()), buff, bufc); } } else { // Traditional single-char: %q0-%q9, %qa-%qz // int i = mux_RegisterSet[static_cast(txt[2])]; if ( 0 <= i && i < MAX_GLOBAL_REGS) { if ( mudstate.global_regs[i] && mudstate.global_regs[i]->reg_len > 0) { safe_copy_buf(mudstate.global_regs[i]->reg_ptr, mudstate.global_regs[i]->reg_len, buff, bufc); } } } } break; case '#': // %# — enactor dbref. // { scratch[0] = '#'; size_t n = mux_ltoa(enactor, scratch + 1); safe_copy_buf(scratch, n + 1, buff, bufc); } break; case '!': // %! — executor dbref. // { scratch[0] = '#'; size_t n = mux_ltoa(executor, scratch + 1); safe_copy_buf(scratch, n + 1, buff, bufc); } break; case '@': // %@ — caller dbref. // { scratch[0] = '#'; size_t n = mux_ltoa(caller, scratch + 1); safe_copy_buf(scratch, n + 1, buff, bufc); } break; case '%': // %% — literal percent. // safe_chr('%', buff, bufc); break; case 'R': // %r — carriage return. // safe_copy_buf(T("\r\n"), 2, buff, bufc); break; case 'B': // %b — blank (space). // safe_chr(' ', buff, bufc); break; case 'T': // %t — tab. // safe_chr('\t', buff, bufc); break; case 'N': // %n/%N — enactor name. // safe_str(Name(enactor), buff, bufc); break; case 'L': // %l — enactor location dbref. // if (!(eval & EV_NO_LOCATION)) { scratch[0] = '#'; size_t n = mux_ltoa(where_is(enactor), scratch + 1); safe_copy_buf(scratch, n + 1, buff, bufc); } break; case 'S': // %s/%S — subjective pronoun. // { const PRONOUN_SET *ps = get_pronoun_set(enactor); safe_str(ps->subjective, buff, bufc); } break; case 'P': // %p/%P — possessive pronoun. // { const PRONOUN_SET *ps = get_pronoun_set(enactor); safe_str(ps->possessive, buff, bufc); } break; case 'O': // %o/%O — objective pronoun. // { const PRONOUN_SET *ps = get_pronoun_set(enactor); safe_str(ps->objective, buff, bufc); } break; case 'A': // %a/%A — absolute possessive pronoun. // { const PRONOUN_SET *ps = get_pronoun_set(enactor); safe_str(ps->absolute, buff, bufc); } break; case 'M': // %m — last command. // safe_str(mudstate.curr_cmd, buff, bufc); break; case 'K': // %k — moniker. // safe_str(Moniker(enactor), buff, bufc); break; case '|': // %| — piped command output. // safe_str(mudstate.pout, buff, bufc); break; case '+': // %+ — number of command args. // safe_i64toa(ncargs, buff, bufc); break; case ':': // %: — enactor objid (#dbref:creation_seconds). // { scratch[0] = '#'; size_t n = mux_ltoa(enactor, scratch + 1); int64_t csecs = creation_seconds(enactor); if (0 != csecs) { scratch[n + 1] = ':'; mux_i64toa(csecs, scratch + n + 2); } safe_str(scratch, buff, bufc); } break; case 'V': // %va-%vz — variable attribute. // if (txt.size() >= 3 && mux_isazAZ(txt[2])) { int i = A_VA + mux_toupper_ascii(txt[2]) - 'A'; dbref aowner; int aflags; size_t nAttrGotten; atr_pget_str_LEN(scratch, executor, i, &aowner, &aflags, &nAttrGotten); if (0 < nAttrGotten) { safe_copy_buf(scratch, nAttrGotten, buff, bufc); } } break; case 'I': // %i0-%i9 — itext() substitution. // if (txt.size() >= 3 && mux_isdigit(txt[2])) { int depth = txt[2] - '0'; int i = mudstate.in_loop - depth - 1; if (0 <= i && i < MAX_ITEXT) { safe_str(mudstate.itext[i], buff, bufc); } } else if (txt.size() >= 3) { // %i followed by non-digit — output the char after %i literally. // safe_chr(txt[2], buff, bufc); } break; case '=': // %= — plain equals sign. // %= — attribute or numbered arg substitution. // if (txt.size() >= 4 && txt[2] == '<') { size_t close = txt.find('>', 3); if (close != std::string::npos) { size_t nName = close - 3; memcpy(scratch, txt.c_str() + 3, nName); scratch[nName] = '\0'; if (mux_isdigit(scratch[0])) { // Numeric arg reference: %=<0> through %=<999> // int i; if (!mux_isdigit(scratch[1])) { i = scratch[0] - '0'; } else if (!mux_isdigit(scratch[2])) { i = TableATOI(scratch[0] - '0', scratch[1] - '0'); } else if (!mux_isdigit(scratch[3])) { i = 10 * TableATOI(scratch[0] - '0', scratch[1] - '0') + scratch[2] - '0'; } else { i = MAX_ARG; } if (i < ncargs && nullptr != cargs[i]) { safe_str(cargs[i], buff, bufc); } } else if (mux_isattrnameinitial(scratch)) { ATTR *ap = atr_str(scratch); if (ap && See_attr(executor, executor, ap)) { dbref aowner; int aflags; size_t nLen; atr_pget_str_LEN(scratch, executor, ap->number, &aowner, &aflags, &nLen); safe_copy_buf(scratch, nLen, buff, bufc); } } } else { // Malformed %= — output literally. // safe_str(reinterpret_cast(txt.c_str()), buff, bufc); } } break; case 'C': case 'X': // %c/%x — color codes. // Uppercase C/X → background (0x40 flag in L2 table). // Lowercase c/x → foreground. // if (txt.size() >= 3) { bool bBackground = ('A' <= ch && ch <= 'Z'); if (txt[2] == '<') { // Extended color: %c, %x, etc. // size_t close = txt.find('>', 3); if (close != std::string::npos) { size_t nColor = close - 3; const UTF8 *pColor = reinterpret_cast(txt.c_str() + 3); RGB rgb; if (parse_rgb(nColor, pColor, rgb)) { // Emit via LettersToBinary — the same // ColorTransitionBinary / EmitSMPColor (v5 two- // codepoint) path ansi() uses. The previous // hand-rolled 0xF0x00+channel scheme was the // retired v4 three-codepoint delta encoding; // the live decoder is v5, so non-palette-exact // truecolor was corrupted (#1933). // // Letter form: "" FG, "/" BG. // Body length is tiny (#RRGGBB or "R G B"). // UTF8 letters[32]; size_t li = 0; if (bBackground) { letters[li++] = '/'; } letters[li++] = '<'; if (li + nColor + 1 < sizeof(letters)) { memcpy(letters + li, pColor, nColor); li += nColor; letters[li++] = '>'; letters[li] = '\0'; safe_str(LettersToBinary(letters), buff, bufc); } } } else { // Malformed %c< color with no closing > — output literally. // safe_str(reinterpret_cast(txt.c_str()), buff, bufc); } } else { // Simple color code: %xn, %ch, etc. // unsigned int iColor = ColorTable[static_cast(txt[2])]; if (iColor) { safe_str(aColors[iColor].pUTF, buff, bufc); } else { // Unknown color letter — output it literally. // safe_chr(txt[2], buff, bufc); } } } break; default: // Unknown substitution — output the character literally // (matches iCode == 0 in mux_exec). // safe_chr(ch, buff, bufc); break; } // For uppercase escape letters (%S, %N, %P, %O, %A, %K), // uppercase the first character of the substituted value. // if (bUpperCase) { mux_toupper_first(TempPtr, bufc, LBUF_SIZE); } } // --------------------------------------------------------------- // Native NOEVAL handlers // --------------------------------------------------------------- // // These functions handle specific NOEVAL built-in functions by // evaluating AST subtrees directly instead of serializing back // to text and re-parsing through mux_exec. // // Helper: evaluate an AST subtree for a NOEVAL branch (if/switch/iter). // // ##/#@/#$ are resolved natively as AST_SUBST nodes at eval time // (from mudstate.itext/inum/switch_token). // static void ast_eval_branch(const ASTNode *callNode, int childIndex, const ASTNode *child, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { const ASTNode *noevalNode = nullptr; const std::string *rawText = nullptr; std::string fallbackRaw; if (ast_call_arg_is_deferred(callNode, childIndex)) { noevalNode = child; rawText = ast_call_raw_arg(callNode, childIndex); } else if (child) { fallbackRaw = ast_raw_text(child); rawText = &fallbackRaw; } ast_eval_deferred_region(child, noevalNode, rawText, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } // Native cand/candbool: short-circuit AND. // static void ast_noeval_cand(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs, bool bBool) { int nfargs = static_cast(node->children.size()); bool val = true; LBuf temp = LBuf_Src("ast_noeval_cand"); for (int i = 0; i < nfargs && val && !alarm_clock.alarmed; i++) { UTF8 *bp = temp; ast_eval_node(node->children[i].get(), temp, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; val = bBool ? xlate(temp) : isTRUE(mux_atoi64(temp)); } safe_bool(val, buff, bufc); } // Native cor/corbool: short-circuit OR. // static void ast_noeval_cor(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs, bool bBool) { int nfargs = static_cast(node->children.size()); bool val = false; LBuf temp = LBuf_Src("ast_noeval_cor"); for (int i = 0; i < nfargs && !val && !alarm_clock.alarmed; i++) { UTF8 *bp = temp; ast_eval_node(node->children[i].get(), temp, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; val = bBool ? xlate(temp) : isTRUE(mux_atoi64(temp)); } safe_bool(val, buff, bufc); } // Native if/ifelse: conditional branch selection. // static void ast_noeval_ifelse(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { int nfargs = static_cast(node->children.size()); // Evaluate the condition. // LBuf lbuff = LBuf_Src("ast_noeval_if"); UTF8 *bp = lbuff; ast_eval_node(node->children[0].get(), lbuff, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; const UTF8 *saved_switch = mudstate.switch_token; mudstate.switch_token = lbuff; if (xlate(lbuff)) { ast_eval_branch(node, 1, node->children[1].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } else if (nfargs >= 3) { ast_eval_branch(node, 2, node->children[2].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } mudstate.switch_token = saved_switch; } // Native switch/case: first-match pattern dispatch. // static void ast_noeval_switch(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs, bool bWild) { int nfargs = static_cast(node->children.size()); // Evaluate the target in child[0]. // LBuf mbuff = LBuf_Src("ast_noeval_switch"); UTF8 *bp = mbuff; ast_eval_node(node->children[0].get(), mbuff, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; LBuf tbuff = LBuf_Src("ast_noeval_switch.2"); // Loop through patterns looking for a match. #$ is bound to the // match target only while a matched (or the default) is // evaluated, never while a pattern is evaluated (#857; mirrors // switch_handler in functions.cpp). // int i; for (i = 1; i < nfargs - 1 && !alarm_clock.alarmed; i += 2) { bp = tbuff; ast_eval_node(node->children[i].get(), tbuff, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; if ( bWild ? wild_match(tbuff, mbuff) : strcmp(reinterpret_cast(tbuff.get()), reinterpret_cast(mbuff.get())) == 0) { const UTF8 *saved_switch = mudstate.switch_token; mudstate.switch_token = mbuff; ast_eval_branch(node, i + 1, node->children[i + 1].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); mudstate.switch_token = saved_switch; return; } } // No match — evaluate default if present. // if (i < nfargs) { const UTF8 *saved_switch = mudstate.switch_token; mudstate.switch_token = mbuff; ast_eval_branch(node, i, node->children[i].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); mudstate.switch_token = saved_switch; } } // Native switchall/caseall: all-match pattern dispatch. // static void ast_noeval_switchall(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs, bool bWild) { int nfargs = static_cast(node->children.size()); // Evaluate the target in child[0]. // LBuf mbuff = LBuf_Src("ast_noeval_switchall"); UTF8 *bp = mbuff; ast_eval_node(node->children[0].get(), mbuff, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; LBuf tbuff = LBuf_Src("ast_noeval_switchall.2"); // Loop through all patterns, evaluating every match. As in // ast_noeval_switch, #$ is bound only around /default // evaluation, never pattern evaluation (#857). // bool bMatched = false; int i; for (i = 1; i < nfargs - 1 && !alarm_clock.alarmed; i += 2) { bp = tbuff; ast_eval_node(node->children[i].get(), tbuff, &bp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *bp = '\0'; if ( bWild ? wild_match(tbuff, mbuff) : strcmp(reinterpret_cast(tbuff.get()), reinterpret_cast(mbuff.get())) == 0) { bMatched = true; const UTF8 *saved_switch = mudstate.switch_token; mudstate.switch_token = mbuff; ast_eval_branch(node, i + 1, node->children[i + 1].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); mudstate.switch_token = saved_switch; } } // If nothing matched, evaluate the default. // if (!bMatched && i < nfargs) { const UTF8 *saved_switch = mudstate.switch_token; mudstate.switch_token = mbuff; ast_eval_branch(node, i, node->children[i].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); mudstate.switch_token = saved_switch; } } // Native iter: list iteration. // static void ast_noeval_iter(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { int nfargs = static_cast(node->children.size()); // Handle optional delimiters (args 3 and 4) by serializing // and evaluating them, then parsing into SEP structures. // For the common case (no delimiters), use space defaults. // SEP sep; sep.n = 1; memcpy(sep.str, " ", 2); SEP osep; osep.n = 1; memcpy(osep.str, " ", 2); if (nfargs >= 3) { // Evaluate input delimiter. // LBuf dbuf = LBuf_Src("ast_noeval_iter.sep"); UTF8 *dp = dbuf; ast_eval_node(node->children[2].get(), dbuf, &dp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *dp = '\0'; size_t dlen = dp - dbuf; if (dlen == 1) { sep.n = 1; memcpy(sep.str, dbuf.get(), 2); } else if (dlen > 1 && dlen <= MAX_SEP_LEN) { sep.n = dlen; memcpy(sep.str, dbuf.get(), dlen); sep.str[dlen] = '\0'; } } if (nfargs >= 4) { // Evaluate output delimiter. // LBuf dbuf = LBuf_Src("ast_noeval_iter.osep"); UTF8 *dp = dbuf; ast_eval_node(node->children[3].get(), dbuf, &dp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *dp = '\0'; size_t dlen = dp - dbuf; if (dlen == 0) { osep.n = 1; memcpy(osep.str, " ", 2); } else if (dlen == 2 && memcmp(dbuf.get(), "@@", 2) == 0) { osep.n = 0; osep.str[0] = '\0'; } else if (dlen == 2 && memcmp(dbuf.get(), "\r\n", 2) == 0) { osep.n = 2; memcpy(osep.str, "\r\n", 3); } else if (dlen == 1) { osep.n = 1; memcpy(osep.str, dbuf.get(), 2); } else if (dlen <= MAX_SEP_LEN) { osep.n = dlen; memcpy(osep.str, dbuf.get(), dlen); osep.str[dlen] = '\0'; } } // Evaluate the list (child[0]). // LBuf curr = LBuf_Src("ast_noeval_iter"); UTF8 *dp = curr; ast_eval_node(node->children[0].get(), curr, &dp, executor, caller, enactor, eval|EV_STRIP_CURLY|EV_FCHECK|EV_EVAL, cargs, ncargs); *dp = '\0'; size_t ncp; UTF8 *cp = trim_space_sep_LEN(curr, dp - curr, sep, &ncp); if (!*cp) { return; } bool first = true; int number = 0; bool bLoopInBounds = ( 0 <= mudstate.in_loop && mudstate.in_loop < MAX_ITEXT); if (bLoopInBounds) { mudstate.itext[mudstate.in_loop] = nullptr; mudstate.inum[mudstate.in_loop] = number; } mudstate.in_loop++; while ( cp && mudstate.func_invk_ctr < mudconf.func_invk_lim && !alarm_clock.alarmed) { if (!first) { print_sep(osep, buff, bufc); } first = false; number++; UTF8 *objstring = split_token(&cp, sep); if (bLoopInBounds) { mudstate.itext[mudstate.in_loop - 1] = objstring; mudstate.inum[mudstate.in_loop - 1] = number; } // iter() body is collected through the FN_NOEVAL arg path and // then re-evaluated per item. Preserve that boundary by // routing through the deferred branch helper as well. ast_eval_branch(node, 1, node->children[1].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } mudstate.in_loop--; if (bLoopInBounds) { mudstate.itext[mudstate.in_loop] = nullptr; mudstate.inum[mudstate.in_loop] = 0; } } // ulambda(body, arg0, arg1, ...): anonymous function evaluation. // // The first argument (body) is received UNEVALUATED — the parser's // deferred-eval mechanism prevents inner function calls from being // dispatched. We reconstruct the raw text, then evaluate it with // the remaining args as %0-%9 substitutions. // // The body can be: // - #lambda/code → extract code after "#lambda/" // - #apply[N]/func → synthesize func(%0,...,%N-1) // - obj/attr → resolve attribute from object // - attrname → resolve attribute from executor // static void ast_noeval_ulambda(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { UNUSED_PARAMETER(caller); int nfargs = static_cast(node->children.size()); if (nfargs < 1) return; // Reconstruct the first argument from the raw deferred text. // The deferred mechanism preserves the raw token text, bypassing // function-call parsing of the body. We evaluate it with // EV_STRIP_CURLY|EV_EVAL but WITHOUT EV_FCHECK so inner function // names like mul(...) are emitted as literal text, not dispatched. // LBuf raw_arg0 = LBuf_Src("ulambda.arg0"); UTF8 *rp = raw_arg0; const std::string *rawText = ast_call_raw_arg(node, 0); if (rawText && !rawText->empty()) { UTF8 *rawCopy = alloc_lbuf("ulambda.raw"); size_t rawLen = rawText->size(); if (rawLen > LBUF_SIZE - 1) rawLen = LBUF_SIZE - 1; memcpy(rawCopy, rawText->c_str(), rawLen); rawCopy[rawLen] = '\0'; mux_exec(rawCopy, rawLen, raw_arg0, &rp, executor, executor, enactor, EV_STRIP_CURLY | EV_EVAL, cargs, ncargs); free_lbuf(rawCopy); } *rp = '\0'; // Evaluate the remaining arguments (these are NOT deferred). // UTF8 *fargs[MAX_ARG]; memset(fargs, 0, sizeof(fargs)); fargs[0] = raw_arg0; int real_nfargs = nfargs; if (real_nfargs > MAX_ARG) real_nfargs = MAX_ARG; for (int i = 1; i < real_nfargs; i++) { fargs[i] = alloc_lbuf("ulambda.arg"); UTF8 *bp = fargs[i]; ast_eval_node(node->children[i].get(), fargs[i], &bp, executor, executor, enactor, eval | EV_FCHECK | EV_EVAL, cargs, ncargs); *bp = '\0'; } // Use parse_and_get_attrib to handle #lambda/, #apply/, obj/attr. // UTF8 *atext; dbref thing; dbref aowner; int aflags; if (!parse_and_get_attrib(executor, fargs, &atext, &thing, &aowner, &aflags, buff, bufc)) { for (int i = 1; i < real_nfargs; i++) free_lbuf(fargs[i]); return; } // Mirror do_ufun (fun_u): a NO_EVAL attribute or a NO_EVAL object // returns the attribute text literally instead of evaluating it // (#786). This branch was the one piece of do_ufun the ulambda // route was missing. // if ((aflags & AF_NOEVAL) || NoEval(thing)) { size_t nLen = strlen(reinterpret_cast(atext)); safe_copy_buf(atext, nLen, buff, bufc); } else { // Evaluate the body with fargs[1]... as %0, %1, ... // // Use mux_exec so the body JIT-compiles, mirroring do_ufun // (fun_u). The cargs we pass here (&fargs[1]) populate %0-%9 // correctly: a depth-1 run_cached_program reads them from // CARGS_BASE. This was previously routed through ast_exec to // dodge a blob-internal float intrinsic bug (#778); now that // #778 is fixed, the JIT path is safe. // mux_exec(atext, LBUF_SIZE - 1, buff, bufc, thing, executor, enactor, AttrTrace(aflags, EV_FCHECK | EV_EVAL), const_cast(&fargs[1]), real_nfargs - 1); } free_lbuf(atext); for (int i = 1; i < real_nfargs; i++) free_lbuf(fargs[i]); } // Dispatch table for native NOEVAL handling. Returns true if the // function was handled natively (caller should skip generic dispatch). // static bool ast_try_native_noeval(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if (!node) { return false; } switch (node->noeval_kind) { case ASTNOEVAL_IFELSE: ast_noeval_ifelse(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return true; case ASTNOEVAL_SWITCH: ast_noeval_switch(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, true); return true; case ASTNOEVAL_CASE: ast_noeval_switch(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, false); return true; case ASTNOEVAL_SWITCHALL: ast_noeval_switchall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, true); return true; case ASTNOEVAL_CASEALL: ast_noeval_switchall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, false); return true; case ASTNOEVAL_ITER: ast_noeval_iter(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return true; case ASTNOEVAL_CAND: ast_noeval_cand(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, false); return true; case ASTNOEVAL_CANDBOOL: ast_noeval_cand(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, true); return true; case ASTNOEVAL_COR: ast_noeval_cor(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, false); return true; case ASTNOEVAL_CORBOOL: ast_noeval_cor(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs, true); return true; case ASTNOEVAL_ULAMBDA: ast_noeval_ulambda(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return true; case ASTNOEVAL_NONE: default: return false; } } static std::string ast_raw_arg_text(const ASTNode *callNode, int argIndex) { if (!callNode || argIndex < 0) { return ""; } if (const std::string *raw = ast_call_raw_arg(callNode, argIndex)) { return *raw; } if (argIndex < static_cast(callNode->children.size())) { return ast_raw_text(callNode->children[argIndex].get()); } return ""; } // Output a function call node as literal text: name(arg,arg,...). // static void ast_emit_literal_funccall(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); safe_chr('(', buff, bufc); // The function name did not resolve — this is not a real function call. // Strip EV_FCHECK so that nested FUNCCALL children are emitted as // literal text, matching the classic evaluator's behavior where a // failed function-call check consumes the FCHECK opportunity. // int childEval = eval & ~EV_FCHECK; for (size_t i = 0; i < node->children.size(); i++) { if (i > 0) safe_chr(',', buff, bufc); ast_eval_node(node->children[i].get(), buff, bufc, executor, caller, enactor, childEval, cargs, ncargs); } safe_chr(')', buff, bufc); } // Evaluate a function call node (AST_FUNCCALL). // static void ast_eval_funccall(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { // EV_FCHECK without EV_FMAND means "check if the first ( is a // function call." If EV_FCHECK has been stripped (by the SEQUENCE // handler after the first child), this function call is not at // the start of the expression — output as literal text. // if (!(eval & EV_FCHECK) && !(eval & EV_FMAND)) { ast_emit_literal_funccall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return; } // Missing closing ')' — the function is NOT dispatched. Output // the function name and '(' literally, then evaluate the argument // children (%-substitutions must be resolved, matching the classic // parser which evaluates as it scans). // if (!node->has_close_paren) { safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); safe_chr('(', buff, bufc); for (size_t i = 0; i < node->children.size(); i++) { if (i > 0) safe_chr(',', buff, bufc); ast_eval_node(node->children[i].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } return; } // Uppercase the function name for lookup. // size_t nName = node->text.size(); if (nName == 0 || nName > MAX_UFUN_NAME_LEN) { ast_emit_literal_funccall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); return; } LBuf TempFun = LBuf_Src("ast_eval_funccall"); memcpy(TempFun, node->text.c_str(), nName); TempFun[nName] = '\0'; size_t nUpper; UTF8 *pUpper = mux_strupr(TempFun, nUpper); if (nUpper >= LBUF_SIZE) { nUpper = LBUF_SIZE - 1; } memcpy(TempFun, pUpper, nUpper); TempFun[nUpper] = '\0'; std::vector name_key(TempFun.get(), TempFun.get() + nUpper); FUN *fp = nullptr; UFUN *ufp = nullptr; const auto it = mudstate.builtin_functions.find(name_key); if (it != mudstate.builtin_functions.end()) { fp = it->second; } if (!fp) { auto it_ufunc = mudstate.ufunc_htab.find(name_key); ufp = (it_ufunc != mudstate.ufunc_htab.end()) ? static_cast(it_ufunc->second) : nullptr; } if (!fp && !ufp) { if (eval & EV_FMAND) { safe_str(S_("#-1 FUNCTION ("), buff, bufc); safe_str(TempFun, buff, bufc); safe_str(T(") NOT FOUND"), buff, bufc); s_fmand_abort = true; } else { ast_emit_literal_funccall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } return; } // Check invocation limits. // mudstate.func_nest_lev++; mudstate.func_invk_ctr++; UTF8 *oldp = *bufc; if (mudconf.func_nest_lim <= mudstate.func_nest_lev) { safe_str(S_("#-1 FUNCTION RECURSION LIMIT EXCEEDED"), buff, bufc); } else if (mudconf.func_invk_lim <= mudstate.func_invk_ctr) { safe_str(S_("#-1 FUNCTION INVOCATION LIMIT EXCEEDED"), buff, bufc); } else if (Going(executor)) { safe_str(S_("#-1 BAD EXECUTOR"), buff, bufc); } else if (!check_access(executor, ufp ? ufp->perms : fp->perms)) { safe_noperm(buff, bufc); } else if (ufp && (ufp->flags & FN_RESTRICT) && !Wizard(executor)) { safe_noperm(buff, bufc); } else if (alarm_clock.alarmed) { safe_str(S_("#-1 CPU LIMITED"), buff, bufc); } else if (ufp) { // User-defined function — fetch attribute and evaluate. // dbref aowner; int aflags; UTF8 *tbuf = atr_get("ast_eval.ufun", ufp->obj, ufp->atr, &aowner, &aflags); dbref obj = (ufp->flags & FN_PRIV) ? ufp->obj : executor; int nfargs = static_cast(node->children.size()); if (nfargs > MAX_ARG) { nfargs = MAX_ARG; } // Evaluate arguments for UFUN. // UTF8 *fargs[MAX_ARG]; memset(fargs, 0, sizeof(fargs)); for (int i = 0; i < nfargs; i++) { fargs[i] = alloc_lbuf("ast_eval.ufun.arg"); UTF8 *bp = fargs[i]; ast_eval_node(node->children[i].get(), fargs[i], &bp, executor, caller, enactor, eval | EV_FCHECK | EV_EVAL, cargs, ncargs); *bp = '\0'; } if ((aflags & AF_NOEVAL) || NoEval(ufp->obj)) { size_t nLen = strlen(reinterpret_cast(tbuf)); safe_copy_buf(tbuf, nLen, buff, bufc); } else { reg_ref **preserve = nullptr; if (ufp->flags & FN_PRES) { preserve = PushRegisters(MAX_GLOBAL_REGS); save_global_regs(preserve); } int feval = eval & ~(EV_TOP | EV_FMAND); mux_exec(tbuf, LBUF_SIZE-1, buff, bufc, obj, executor, enactor, AttrTrace(aflags, feval), const_cast(fargs), nfargs); if (ufp->flags & FN_PRES) { restore_global_regs(preserve); PopRegisters(preserve, MAX_GLOBAL_REGS); } } for (int i = 0; i < nfargs; i++) { free_lbuf(fargs[i]); } free_lbuf(tbuf); } else { // Built-in function. // // The AST parser always splits on commas, but mux_exec uses // maxArgsParsed to limit splitting — excess args are catenated // (with commas) into the last slot. Apply the same logic here. // int nParsed = static_cast(node->children.size()); int nfargs = nParsed; if (nfargs > fp->maxArgsParsed && fp->maxArgsParsed > 0) { nfargs = fp->maxArgsParsed; } if (nfargs > MAX_ARG) { nfargs = MAX_ARG; } if ( fp->minArgs <= nfargs && nfargs <= fp->maxArgs && !alarm_clock.alarmed) { // Try native NOEVAL handlers first. // if ( (fp->flags & FN_NOEVAL) && ast_try_native_noeval(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs)) { mudstate.func_nest_lev--; return; } UTF8 *fargs[MAX_ARG]; memset(fargs, 0, sizeof(fargs)); int feval; if (fp->flags & FN_NOEVAL) { // NOEVAL functions receive raw text. The handler // calls mux_exec internally on args it wants to // evaluate. // feval = eval & ~(EV_EVAL | EV_TOP | EV_FMAND | EV_STRIP_CURLY); for (int i = 0; i < nfargs; i++) { fargs[i] = alloc_lbuf("ast_eval.noeval"); if (i < nfargs - 1 || nParsed <= nfargs) { std::string raw = ast_raw_arg_text(node, i); size_t len = raw.size(); if (len >= LBUF_SIZE) len = LBUF_SIZE - 1; memcpy(fargs[i], raw.c_str(), len); fargs[i][len] = '\0'; } else { // Catenate remaining children with commas. // UTF8 *bp = fargs[i]; for (int j = i; j < nParsed; j++) { if (j > i) safe_chr(',', fargs[i], &bp); std::string raw = ast_raw_arg_text(node, j); safe_str(reinterpret_cast(raw.c_str()), fargs[i], &bp); } *bp = '\0'; } } } else { // Normal functions: evaluate each argument. // feval = eval & ~(EV_TOP | EV_FMAND); for (int i = 0; i < nfargs; i++) { fargs[i] = alloc_lbuf("ast_eval.arg"); UTF8 *bp = fargs[i]; if (i < nfargs - 1 || nParsed <= nfargs) { ast_eval_argument(node->children[i].get(), fargs[i], &bp, executor, caller, enactor, feval | EV_FCHECK | EV_EVAL, cargs, ncargs); } else { // Catenate remaining children with commas. // for (int j = i; j < nParsed; j++) { if (j > i) safe_chr(',', fargs[i], &bp); ast_eval_argument(node->children[j].get(), fargs[i], &bp, executor, caller, enactor, feval | EV_FCHECK | EV_EVAL, cargs, ncargs); } } *bp = '\0'; } } fp->fun(fp, buff, &oldp, executor, caller, enactor, feval & EV_TRACE, fargs, nfargs, cargs, ncargs); *bufc = oldp; for (int i = 0; i < nfargs; i++) { free_lbuf(fargs[i]); } } else { // Wrong argument count. // if (fp->minArgs == fp->maxArgs) { safe_tprintf_str(buff, bufc, S_("#-1 FUNCTION (%s) EXPECTS %d ARGUMENTS"), fp->name, fp->minArgs); } else if (fp->minArgs + 1 == fp->maxArgs) { safe_tprintf_str(buff, bufc, S_("#-1 FUNCTION (%s) EXPECTS %d OR %d ARGUMENTS"), fp->name, fp->minArgs, fp->maxArgs); } else { safe_tprintf_str(buff, bufc, S_("#-1 FUNCTION (%s) EXPECTS BETWEEN %d AND %d ARGUMENTS"), fp->name, fp->minArgs, fp->maxArgs); } } } mudstate.func_nest_lev--; } // Hard cap on AST evaluator C-stack recursion. This is independent of // mudconf.func_nest_lim (function call depth) and mudconf.nStackLimit // (softcode bracket nesting). It exists to keep adversarial or // pathologically deep ASTs — e.g., `[[[[...x]]]]` with thousands of // bracket layers, or deeply chained `if()`/`switch()` constructs — // from blowing the native C stack before the soft limits trip on a // subsequent re-entry into mux_exec. // // Each AST_EVALBRACKET/AST_BRACEGROUP layer adds ~2 frames (bracket → // sequence → inner bracket), so a cap of 400 bounds the stack around // 800 frames — well under a default 8 MiB Linux stack and safe on // platforms with smaller defaults. // static constexpr int AST_EVAL_MAX_DEPTH = 400; static thread_local int s_ast_eval_depth = 0; class AstEvalDepthGuard { public: AstEvalDepthGuard() { s_ast_eval_depth++; } ~AstEvalDepthGuard() { s_ast_eval_depth--; } bool overflow() const { return s_ast_eval_depth > AST_EVAL_MAX_DEPTH; } }; // Evaluate a single AST node into buff/bufc. // static void ast_eval_node(const ASTNode *node, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if (!node || alarm_clock.alarmed) { return; } AstEvalDepthGuard depth_guard; if (depth_guard.overflow()) { mudstate.bStackLimitReached = true; return; } // Softcode execution trace (#1023). Record (source -> result) for the // call boundaries -- function calls and eval brackets -- when tracing is // armed. These are the subexpressions a builder debugging code cares // about; the change-filter in tcache_add drops no-op nodes. Off-path // cost is a single bit test, since EV_TRACE is only set while tracing. // const bool bTrace = (eval & EV_TRACE) && !(eval & EV_NOTRACE) && ( AST_FUNCCALL == node->type || AST_EVALBRACKET == node->type); UTF8 *pTraceStart = bTrace ? *bufc : nullptr; switch (node->type) { case AST_LITERAL: safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); break; case AST_SPACE: if ( mudconf.space_compress && !(eval & EV_NO_COMPRESS)) { // Space compression: emit a single space regardless of // how many whitespace characters are in the source. // safe_chr(' ', buff, bufc); } else { safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); } break; case AST_SEMICOLON: safe_chr(';', buff, bufc); break; case AST_SUBST: if (eval & EV_EVAL) { ast_eval_subst(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } else { // Without EV_EVAL, pass through literally. // safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); } break; case AST_ESCAPE: // Output the escaped character (skip the backslash). // if (node->text.size() > 1) { safe_chr(node->text[1], buff, bufc); } break; case AST_FUNCCALL: if (eval & EV_FCHECK) { ast_eval_funccall(node, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } else { // Without EV_FCHECK, output as literal text. // safe_str(reinterpret_cast(node->text.c_str()), buff, bufc); safe_chr('(', buff, bufc); for (size_t i = 0; i < node->children.size(); i++) { if (i > 0) safe_chr(',', buff, bufc); ast_eval_node(node->children[i].get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } safe_chr(')', buff, bufc); } break; case AST_EVALBRACKET: if (eval & EV_NOFCHECK) { // Brackets suppressed — pass through as literal text. // safe_chr('[', buff, bufc); if (!node->children.empty()) { std::string raw = ast_raw_text(node->children[0].get()); safe_str(reinterpret_cast(raw.c_str()), buff, bufc); } safe_chr(']', buff, bufc); } else { // Evaluate contents with function checking enabled. // mudstate.nStackNest++; if (!node->children.empty()) { // Confine the abort signal to this region (#1247): an // inner [...] that fails must not truncate the region // containing it, and neither must a failure inside an // attribute reached from here. // bool saved_abort = s_fmand_abort; s_fmand_abort = false; ast_eval_node(node->children[0].get(), buff, bufc, executor, caller, enactor, eval | EV_FCHECK | EV_FMAND, cargs, ncargs); s_fmand_abort = saved_abort; } mudstate.nStackNest--; } break; case AST_BRACEGROUP: mudstate.nStackNest++; if (eval & EV_STRIP_CURLY) { // Strip braces and evaluate contents without // function checking. // int innerEval = eval & ~(EV_STRIP_CURLY | EV_FCHECK | EV_FMAND); if (!node->children.empty()) { ast_eval_node(node->children[0].get(), buff, bufc, executor, caller, enactor, innerEval, cargs, ncargs); } } else { // Pass through as literal braces. // safe_chr('{', buff, bufc); if (!node->children.empty()) { int innerEval = eval & ~(EV_TOP | EV_FMAND); if (eval & EV_EVAL) { innerEval = innerEval & ~(EV_STRIP_CURLY | EV_FCHECK | EV_FMAND); ast_eval_node(node->children[0].get(), buff, bufc, executor, caller, enactor, innerEval, cargs, ncargs); } else { innerEval = (innerEval & ~EV_FCHECK) | EV_NOFCHECK; ast_eval_node(node->children[0].get(), buff, bufc, executor, caller, enactor, innerEval, cargs, ncargs); } } safe_chr('}', buff, bufc); } mudstate.nStackNest--; break; case AST_SEQUENCE: { size_t first = 0; size_t count = node->children.size(); size_t last = count; if ( mudconf.space_compress && !(eval & EV_NO_COMPRESS)) { while ( last > first && ast_is_malformed_qsubst(node->children[last - 1].get())) { last--; } // Skip leading and trailing AST_SPACE children. // This matches mux_exec's at_space=1 (suppress leading) // and trailing-space strip behavior, without touching // spaces generated by function output. // while (first < count && node->children[first]->type == AST_SPACE) { first++; } while (last > first && node->children[last - 1]->type == AST_SPACE) { last--; } } ast_eval_sequence_children(node, first, last, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); break; } } // Trace emit (#1023): record this call boundary's (source -> result). // tcache_add takes ownership of pOrig and applies the change-filter. // if (bTrace) { std::string src = ast_raw_text(node); UTF8 *pOrig = alloc_lbuf("ast_trace.orig"); size_t nSrc = src.size(); if (nSrc > LBUF_SIZE - 1) { nSrc = LBUF_SIZE - 1; } memcpy(pOrig, src.data(), nSrc); pOrig[nSrc] = '\0'; // The result span [pTraceStart, *bufc) is not NUL-terminated yet. // UTF8 chSaved = **bufc; **bufc = '\0'; tcache_add(executor, pOrig, pTraceStart); **bufc = chSaved; // Bottom-up mode flushes each line as its subexpression completes // (innermost first); top-down accumulates and flushes at the outermost // eval (mux_exec). // if (!mudconf.trace_topdown) { tcache_finish(); } } } void ast_exec(const UTF8 *pStr, size_t nStr, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if ( nullptr == pStr || '\0' == pStr[0] || alarm_clock.alarmed) { return; } size_t nLen = strlen(reinterpret_cast(pStr)); if (nLen > nStr) { nLen = nStr; } auto ast = ast_parse_string(pStr, nLen); if (!ast) { return; } ast_eval_node(ast.get(), buff, bufc, executor, caller, enactor, eval, cargs, ncargs); } // --------------------------------------------------------------- // AST parse cache // --------------------------------------------------------------- // // LRU cache of parsed AST trees, keyed by expression text. // Parsing is pure (no side effects), so cached ASTs are safe // to share across evaluations with different contexts. // struct ASTCacheEntry { std::shared_ptr ast; std::list::iterator lru_it; }; static std::unordered_map s_astCache; static std::list s_astLru; static const size_t AST_CACHE_MAX = 1024; static const size_t AST_CACHE_MIN_LEN = 16; // --------------------------------------------------------------- // Stamp for the memoized jit_can_handle() verdict (#2068). // // The verdict is NOT a pure function of the AST, which is the trap in // "the tree is cached, so cache the answer". It also depends on: // // 1. mudconf.jit_eval_brackets -- a runtime toggle, and one the smoke // suite deliberately runs both ways. Carried in the low bit rather // than hooked, since it is a plain read. // 2. mudstate.ufunc_htab -- @function and @function/delete. // 3. mudstate.builtin_functions -- module (de)registration, and the // function_alias config directive. // // (2) and (3) bump the epoch. Getting this wrong would fail silently and // in the dangerous direction: a name that did not resolve when the verdict // was taken, and does now, would leave a cached "decline" -- or worse, a // cached "accept" for a name that has since gone away. // // The epoch moves in steps of two so the low bit stays the toggle's. A // stamp of 0 means "never computed", which the epoch never produces. // static uint32_t s_jit_gate_epoch = 2; void jit_gate_note_function_table_change(void) { s_jit_gate_epoch += 2; if (0 == s_jit_gate_epoch) { // Wrapped. Skip the reserved "never computed" value; every live // stamp simply misses once and recomputes. // s_jit_gate_epoch = 2; } #if defined(TINYMUX_JIT) // #2130 decline memo: a bail_noop verdict is pure in the compiled // shape, but registration can change what that shape is on the next // compile. Clear here with the gate epoch so both memos move together // (#2140 review). // jit_decline_memo_invalidate(); #endif } static inline uint32_t jit_gate_stamp_now(void) { return s_jit_gate_epoch | (mudconf.jit_eval_brackets ? 1u : 0u); } // --------------------------------------------------------------- // mux_exec — drop-in replacement for mux_exec // --------------------------------------------------------------- // // Parse into AST (with caching), then evaluate via ast_eval_node. // void mux_exec(const UTF8 *pStr, size_t nStr, UTF8 *buff, UTF8 **bufc, dbref executor, dbref caller, dbref enactor, int eval, const UTF8 *cargs[], int ncargs) { if ( nullptr == pStr || '\0' == pStr[0] || alarm_clock.alarmed) { return; } // Stack limit checking. // if (mudconf.nStackLimit < mudstate.nStackNest) { mudstate.bStackLimitReached = true; return; } // nStr is a buffer-size limit, not the string length. // Use the actual string length for caching and parsing. // size_t nLen = strlen(reinterpret_cast(pStr)); if (nLen > nStr) { nLen = nStr; } // Look up in the parse cache. // const ASTNode *ast_ptr; std::shared_ptr cache_holder; std::unique_ptr parse_holder; if (nLen >= AST_CACHE_MIN_LEN) { std::string key(reinterpret_cast(pStr), nLen); auto it = s_astCache.find(key); if (it != s_astCache.end()) { // Cache hit — move to front of LRU. // cache_holder = it->second.ast; s_astLru.splice(s_astLru.begin(), s_astLru, it->second.lru_it); ast_ptr = cache_holder.get(); } else { // Cache miss — parse and insert. // cache_holder = std::shared_ptr( ast_parse_string(pStr, nLen).release()); ast_ptr = cache_holder.get(); // Evict LRU entries if cache is full. // while (s_astCache.size() >= AST_CACHE_MAX) { s_astCache.erase(s_astLru.back()); s_astLru.pop_back(); } s_astLru.push_front(key); s_astCache[key] = {cache_holder, s_astLru.begin()}; } } else { // Short expressions — parse without caching. // parse_holder = ast_parse_string(pStr, nLen); ast_ptr = parse_holder.get(); } // Softcode execution trace (TRACE flag / EV_TRACE). Reconnects the // trace emitter that the AST/JIT evaluator migration orphaned (#1023). // Two things follow from tracing being armed: the accumulated // (input -> output) lines are bracketed and flushed to the owner at the // outermost evaluation, and the JIT is bypassed for this call -- a // compiled blob has no per-node frames for ast_eval_node's hook to // observe, so traced evaluation must run on the AST interpreter. // const bool is_trace = (Trace(executor) || (eval & EV_TRACE)) && !(eval & EV_NOTRACE); bool is_top = false; if (is_trace) { is_top = tcache_empty(); eval |= EV_TRACE; // propagate to the whole subtree } // Evaluate the AST. // #if defined(TINYMUX_JIT) // Only JIT expressions whose %-substitutions are fully supported. // Currently: %0-%9, %b, %r, %t. Unsupported: %#, %!, %n, %l, %q, // %c, %x, etc. Quick scan: if any % is followed by something // we don't handle, fall back to AST. // // Check if the AST is a pure expression the JIT can compile: // - Must be a single function call, eval-bracket, or sequence of // function calls. NOT mixed literal+command text. // - No unsupported %-substitutions. // auto jit_can_handle_compute = [&]() -> bool { if (!ast_ptr) return false; // The AST root must be a function call, eval bracket, or // sequence of function calls/literals. If it's just a literal // (plain text), the JIT adds no value. if (ast_ptr->type == AST_LITERAL || ast_ptr->type == AST_SPACE) return false; // Conservative parser-parity guard: // - eval brackets have subtle re-evaluation behavior // - unterminated calls must match legacy literal/parsing behavior // // Keep these on the classic evaluator until JIT semantics match // exactly. // // Also require at least one function call somewhere in the tree. // A multi-token expression with no calls (e.g. plain command text // such as "+jobs/select monitor") parses as a SEQUENCE of literal // and space nodes whose root is neither AST_LITERAL nor AST_SPACE, // so the single-node check above does not catch it. Compiling such // pure-literal text adds no value, yet it populates the compile // cache with a constant-folded passthrough whose result does not // round-trip through SQLite persistence (the folded output can // reference the source buffer rather than the persisted string // pool), reloading later as an empty string. Keep all // function-free text on the classic evaluator. bool saw_funccall = false; std::vector work; work.push_back(ast_ptr); while (!work.empty()) { const ASTNode *node = work.back(); work.pop_back(); if (node->type == AST_EVALBRACKET) { // Terminated brackets are JITtable when the toggle is // on: hir_lower evaluates their contents in FMAND // context, and the Phase 2-3 q-register coherence work // (slot resync, tracking save/restore, read // materialization) covers the scoping semantics that // made the original unconditional bail load-bearing. // Unterminated brackets keep legacy literal parsing — // always bail. The EV_NOFCHECK literal-passthrough // mode is gated at the call site // (docs/plan-jit-evalbracket-lift.md, Phase 4). if (!mudconf.jit_eval_brackets || !node->has_close_bracket) { return false; } } if (node->type == AST_FUNCCALL) { saw_funccall = true; if (!node->has_close_paren) { return false; } // Bail if the function name doesn't resolve. Unresolved // names (e.g., "eobject=strmatch" from search eval args) // require the AST's EV_FCHECK stripping in // ast_emit_literal_funccall; the JIT doesn't replicate // that behavior. // std::string upper = node->text; for (auto &ch : upper) { ch = static_cast(toupper( static_cast(ch))); } std::vector name_key(upper.begin(), upper.end()); if (mudstate.builtin_functions.find(name_key) == mudstate.builtin_functions.end() && mudstate.ufunc_htab.find(name_key) == mudstate.ufunc_htab.end()) { return false; } } for (const auto &child : node->children) { if (child) { work.push_back(child.get()); } } } // setq/setr: now supported via ECALL_SETQ write-through // (writes to both SUBST slot and mudstate.global_regs). // All standard %-substitutions and # references are now handled // by the compiler. No scanning needed — the AST parser creates // the correct node types, and the compiler resolves each one // at compile time or via ECALL/SUBST slots at runtime. // Pure literal/space text (no function calls anywhere): leave it on // the classic evaluator. See the saw_funccall note above. if (!saw_funccall) { return false; } return true; }; // Memoized gate (#2068). The walk above is the flat tax this removes: // a worklist allocation per evaluation, plus a std::string, a // std::vector and two hash lookups per call node -- paid on every // evaluation, including ones the JIT declines and ones where it wins. // // The stamp covers everything the verdict depends on beyond the tree // itself; see jit_gate_stamp_now(). // auto jit_can_handle = [&]() -> bool { if (!ast_ptr) { return false; } const uint32_t stamp = jit_gate_stamp_now(); if (ast_ptr->jit_gate_stamp == stamp) { return ast_ptr->jit_gate_verdict; } const bool bVerdict = jit_can_handle_compute(); ast_ptr->jit_gate_stamp = stamp; ast_ptr->jit_gate_verdict = bVerdict; return bVerdict; }; if (nLen >= 8 && (eval & EV_EVAL) // Under EV_NOFCHECK, eval brackets pass through as literal // [...] text (the ast.cpp literal path); the JIT lowerer // always evaluates bracket contents, so NOFCHECK text must // stay on the AST evaluator. && !(eval & EV_NOFCHECK) && !alarm_clock.alarmed && !is_trace // traced evaluation must run on the AST interpreter && jit_can_handle()) { if (jit_eval(pStr, nLen, buff, bufc, executor, caller, enactor, eval, cargs, ncargs)) { return; } } #endif ast_eval_node(ast_ptr, buff, bufc, executor, caller, enactor, eval, cargs, ncargs); // Flush accumulated trace lines to the owner at the outermost eval. // ast_eval_node returns normally on alarm/stack-limit, so this is // always reached for the top-level call. // if (is_top) { // trace_limit caps stored lines in top-down mode; tell the owner how // many were dropped (2.13 parity). Read the overflow before finish // resets the counter. // const int nDropped = mudconf.trace_topdown ? tcache_dropped_count() : 0; tcache_finish(); if (0 < nDropped) { notify(executor, tprintf(T("%d lines of trace output discarded."), nDropped)); } } } // --------------------------------------------------------------- // asteval(): softcode function — evaluate via AST evaluator only. // // Bypasses the JIT entirely. Useful for: // - Benchmarking AST vs JIT performance // - Fallback if users find a JIT parity bug // - Verifying parser correctness // // Usage: think asteval(add(1,2)) → 3 // --------------------------------------------------------------- FUNCTION(fun_asteval) { UNUSED_PARAMETER(fp); UNUSED_PARAMETER(cargs); UNUSED_PARAMETER(ncargs); if (nfargs < 1) { safe_str(S_("#-1 TOO FEW ARGUMENTS"), buff, bufc); return; } const UTF8 *expr = fargs[0]; size_t nLen = strlen(reinterpret_cast(expr)); if (nLen == 0) return; auto ast = ast_parse_string(expr, nLen); if (!ast) return; ast_eval_node(ast.get(), buff, bufc, executor, caller, enactor, eval | EV_FCHECK | EV_EVAL, cargs, ncargs); } // jiteval() — the Phase 0 forced-JIT debug oracle — was retired here // after the Phase 5 default flip (docs/plan-jit-evalbracket-lift.md): // with jit_eval_brackets on by default, the production route reaches // everything the gate bypass existed for, and the q-register oracle // compares production evaluation across two workspaces (toggle on vs // explicitly off) instead. // --------------------------------------------------------------- // astbench(): head-to-head AST vs JIT benchmark. // // astbench(, ) // // Runs the expression through both the AST evaluator and the JIT, // reports microseconds per call for each. Output format: // ast=X.XXus jit=Y.YYus ratio=Z.Zx result= // --------------------------------------------------------------- FUNCTION(fun_astbench) { UNUSED_PARAMETER(fp); UNUSED_PARAMETER(cargs); UNUSED_PARAMETER(ncargs); if (nfargs < 2) { safe_str(S_("#-1 TOO FEW ARGUMENTS"), buff, bufc); return; } const UTF8 *expr = fargs[0]; size_t nLen = strlen(reinterpret_cast(expr)); // Clamp in 64-bit BEFORE narrowing. Narrowing first truncates, so // astbench(expr, 4294967296) became 0 iterations and returned nothing, // and 4294967297 became exactly 1 -- while the cap below claims to be // limiting the caller to 100000. A request far above the cap has to // land ON the cap, not wrap past it (#1402). // int64_t iRequested = mux_atoi64(fargs[1]); if (nLen == 0 || iRequested < 1) return; if (iRequested > 100000) iRequested = 100000; int iterations = static_cast(iRequested); // Parse once (shared by both paths). auto ast = ast_parse_string(expr, nLen); if (!ast) { safe_str(S_("#-1 PARSE FAILED"), buff, bufc); return; } // --- AST benchmark --- #ifdef WIN32 LARGE_INTEGER freq, pc0, pc1; QueryPerformanceFrequency(&freq); QueryPerformanceCounter(&pc0); #else struct timespec t0, t1; clock_gettime(CLOCK_MONOTONIC, &t0); #endif for (int i = 0; i < iterations; i++) { LBuf temp = LBuf_Src("astbench_ast"); UTF8 *tp = temp; ast_eval_node(ast.get(), temp, &tp, executor, caller, enactor, eval | EV_FCHECK | EV_EVAL, nullptr, 0); *tp = '\0'; } #ifdef WIN32 QueryPerformanceCounter(&pc1); double ast_us = (double)(pc1.QuadPart - pc0.QuadPart) * 1e6 / ((double)freq.QuadPart * iterations); #else clock_gettime(CLOCK_MONOTONIC, &t1); double ast_us = ((t1.tv_sec - t0.tv_sec) * 1e6 + (t1.tv_nsec - t0.tv_nsec) / 1e3) / iterations; #endif // --- JIT benchmark --- #if defined(TINYMUX_JIT) // Warm the compile cache, then discard what the warm-up emitted. // // Rewind to where THIS call started writing, not to the base of the // buffer. `buff` belongs to the caller, and whatever is already in it is // not ours to throw away: `think TAG [astbench(add(1,2),50)]` silently // lost TAG, because the rewind landed in front of it (#2060). // UTF8 *entry = *bufc; jit_eval(expr, nLen, buff, bufc, executor, caller, enactor, eval | EV_FMAND | EV_EVAL, nullptr, 0); *bufc = entry; // #2133: jit_eval() returns whether it HANDLED the expression, and this // discarded it. A decline is fast -- the gate walks the tree, finds a // shape it cannot lower, and returns -- so an unhandled expression // reported a spectacular jit= time that was the cost of saying no. // citer() read a flat 2.7us at every N, which is absence rendered as // speed, and the field has a standing warning in CLAUDE.md and // tests/growth/README.md telling people not to trust it. // // Count what actually happened instead. A field that cannot answer // should say so rather than return a number that will be read as one. int nHandled = 0; #ifdef WIN32 QueryPerformanceCounter(&pc0); #else clock_gettime(CLOCK_MONOTONIC, &t0); #endif for (int i = 0; i < iterations; i++) { LBuf temp = LBuf_Src("astbench_jit"); UTF8 *tp = temp; if (jit_eval(expr, nLen, temp, &tp, executor, caller, enactor, eval | EV_FMAND | EV_EVAL, nullptr, 0)) { nHandled++; } } #ifdef WIN32 QueryPerformanceCounter(&pc1); double jit_us = (double)(pc1.QuadPart - pc0.QuadPart) * 1e6 / ((double)freq.QuadPart * iterations); #else clock_gettime(CLOCK_MONOTONIC, &t1); double jit_us = ((t1.tv_sec - t0.tv_sec) * 1e6 + (t1.tv_nsec - t0.tv_nsec) / 1e3) / iterations; #endif #else double jit_us = 0.0; int nHandled = 0; #endif // Get the result value. LBuf result = LBuf_Src("astbench_result"); UTF8 *rp = result; ast_eval_node(ast.get(), result, &rp, executor, caller, enactor, eval | EV_FCHECK | EV_EVAL, nullptr, 0); *rp = '\0'; // The JIT leg only reports a time if the JIT actually ran the expression // every time it was asked (#2133). Three distinct outcomes, three // distinct words -- "declined" and "0.03us" must not look alike: // // handled == iterations a real measurement // handled == 0 declined; jit_us is the cost of the decline // 0 < handled < iters mixed, which is a finding in itself: the // verdict changed mid-run, so neither the time // nor the decline describes the whole loop // const bool bAllHandled = (iterations > 0 && nHandled == iterations); const bool bNoneHandled = (nHandled == 0); double ratio = (bAllHandled && jit_us > 0.001) ? ast_us / jit_us : 0.0; // Format floating-point with libc snprintf. mux_vsnprintf / safe_tprintf // only know integer and string conversions — "%.2f" falls through to // mux_assert(0) and aborts the process (#1382). rvbench() already uses // this pattern for the same reason. // char ast_buf[32]; char jit_buf[32]; char ratio_buf[32]; mux_sprintf(reinterpret_cast(ast_buf), sizeof(ast_buf), T("%.2f"), ast_us); if (bAllHandled) { mux_sprintf(reinterpret_cast(jit_buf), sizeof(jit_buf), T("%.2fus"), jit_us); } else if (bNoneHandled) { mux_strncpy(reinterpret_cast(jit_buf), T("declined"), sizeof(jit_buf) - 1); } else { mux_sprintf(reinterpret_cast(jit_buf), sizeof(jit_buf), T("mixed(%d/%d)"), nHandled, iterations); } if (bAllHandled) { mux_sprintf(reinterpret_cast(ratio_buf), sizeof(ratio_buf), T("%.1fx"), ratio); } else { mux_strncpy(reinterpret_cast(ratio_buf), T("n/a"), sizeof(ratio_buf) - 1); } safe_tprintf_str(buff, bufc, T("ast=%sus jit=%s ratio=%s result=%s"), ast_buf, jit_buf, ratio_buf, result.get()); }