From 437fd30313f02fa06aa68c22afcc4bb99f4e6362 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Fri, 25 Sep 2026 01:44:36 +0000 Subject: [PATCH] fix: release FIND expression temporaries --- docs/CURRENT_STATUS.md | 2 + src/claro.c | 2 +- tools/validate_find_expression_cleanup.py | 51 +++++++++++++++++++++++ 3 files changed, 54 insertions(+), 1 deletion(-) create mode 100644 tools/validate_find_expression_cleanup.py diff --git a/docs/CURRENT_STATUS.md b/docs/CURRENT_STATUS.md index 6c7a222..d974f0c 100644 --- a/docs/CURRENT_STATUS.md +++ b/docs/CURRENT_STATUS.md @@ -30,6 +30,8 @@ Verified in this checkout on 2026-09-23: The memory cleanup slices release the previous deep value when a runtime variable or map entry is overwritten, release temporary split argument arrays and strings from `DO`, `CALL`, `TEXT ... CONTAINS`, and `RANDOM`, release loaded program paths, lines, and pointer arrays at the end of each script run, and now release the remaining runtime-owned variables, functions, modules, classes, import paths, captured output, return value, and error strings before the interpreter exits. The expression evaluator now releases discarded intermediate `Value` operands and command boundaries release evaluated `SET`/`SAY` values after copying or printing them. `FOR EACH` now releases its copied collection after iteration, preventing discarded list/map copies from accumulating in long scripts. `IF` and `DO ... TIMES` now release their temporary control-expression values after branch/loop selection. `ASK` now releases evaluated prompt values and temporary input values after converting/copying them into runtime storage. Focused coverage is `tools/validate_memory_cleanup.py`, `tools/validate_expression_cleanup.py`, `tools/validate_control_flow_cleanup.py`, `tools/validate_control_expression_cleanup.py`, and `tools/validate_ask_prompt_cleanup.py`; each focused cleanup validator runs an ASan/UBSan build with LeakSanitizer checking. Verified on 2026-09-24: `python3 tools/validate_ask_prompt_cleanup.py` passes (2,000 prompt/input operations under ASan/UBSan/LSan); `make -s all`, `./claro test`, `./claro doctor`, and `./claro validate` all pass. A malformed-input ASan/UBSan/LSan smoke run did not pass: it reports two leaked `rt_error` message strings (140 bytes total) at `src/claro.c:142`. This unrelated existing diagnostic-path leak remains unaddressed and blocks claiming sanitizer-clean malformed-input handling. The `COUNT ... AS` command now releases its evaluated list/map value after extracting the item count. `python3 tools/validate_count_expression_cleanup.py` first reproduced a 28,000-byte leak across 2,000 list expressions, then passed with LeakSanitizer enabled after the fix. `GET ... AT ... AS ...` now releases its copied collection and empty-result temporary after storing the selected item; `python3 tools/validate_get_expression_cleanup.py` first reproduced 34,000 bytes of leaks across 2,000 empty-list lookups, then passed with LeakSanitizer enabled after the fix. Remaining memory-growth areas include other expression temporaries and command boundaries. The `RUN COMMAND` path remains trusted shell execution, not a sandbox. +The `FIND ... IN ... AS ...` command now releases its evaluated search value and copied list/map after storing the result. Focused coverage is `tools/validate_find_expression_cleanup.py`; it runs 2,000 searches under ASan/UBSan with LeakSanitizer enabled. Verified on 2026-09-24: `python3 tools/validate_find_expression_cleanup.py` passes with no reported leaks; `make -s all`, `./claro test` (0 failures), `./claro doctor`, `./claro validate`, and `git diff --check` pass. This slice is runtime-verified under sanitizers; broader malformed-input sanitizer coverage remains blocked by the previously documented `rt_error` message leak. + The `RUN COMMAND` path now decodes POSIX `pclose()` wait status before storing `LASTEXIT`, so a child that exits with code 3 exposes `3` rather than the encoded status `768`. Focused coverage is `tests/42_last_exit_code.claro`. HTTP responses now have a 1,048,576-byte cap and marker-like response bodies are preserved while extracting the final HTTP status marker. Focused coverage is `tools/validate_http_hardening.py`. Remaining memory-growth areas include other expression temporaries. Claro remains a trusted-script interpreter, not a sandbox. `RUN COMMAND` is documented and validated as a trusted-code capability: it executes shell commands with the user's permissions and is not a sandbox. Claro does not claim untrusted-script safety or use a fragile blacklist sanitizer. Focused documentation coverage is `tools/validate_trusted_command_docs.py`. diff --git a/src/claro.c b/src/claro.c index 9e9e4e6..263f5fe 100644 --- a/src/claro.c +++ b/src/claro.c @@ -459,7 +459,7 @@ static void exec_line(Runtime *rt,Program *p,int *pcp,const char *raw){ char buf if(strcmp(up,"RUN")==0){ const char *as=find_word_ci(t,"AS"); if(as&&starts_ci(t+3," COMMAND")){ char *ex=substr(t+11,as); char *var=xstrdup(trim_inplace((char*)as+2)); Value cv=eval_expr(rt,ex); char *cmd=v_to_string(cv); int code=0; char *out=capture_command_output(cmd,&code,0); rt_set(rt,var,v_str(out)); rt_set(rt,"LASTEXIT",v_num(code)); free(out); free(cmd); free(ex); free(var); } return; } if(strcmp(up,"SORT")==0){ char *name=xstrdup(trim_inplace(t+4)); Value v=rt_get(rt,name); value_list_sort(v); rt_set(rt,name,v); free(name); return; } if(strcmp(up,"REVERSE")==0){ char *name=xstrdup(trim_inplace(t+7)); Value v=rt_get(rt,name); value_list_reverse(v); rt_set(rt,name,v); free(name); return; } - if(strcmp(up,"FIND")==0){ const char *in=find_word_ci(t,"IN"), *as=find_word_ci(t,"AS"); if(in&&as){ char *needle=substr(t+4,in); char *listname=substr(in+2,as); char *var=xstrdup(trim_inplace((char*)as+2)); Value nv=eval_expr(rt,needle), lv=eval_expr(rt,listname); int found=0,i; if(lv.type==V_LIST){ for(i=0;icount;i++) if(value_compare(nv,lv.list->items[i])==0){ found=i+1; break; } } rt_set(rt,var,v_num(found)); free(needle); free(listname); free(var); } return; } + if(strcmp(up,"FIND")==0){ const char *in=find_word_ci(t,"IN"), *as=find_word_ci(t,"AS"); if(in&&as){ char *needle=substr(t+4,in); char *listname=substr(in+2,as); char *var=xstrdup(trim_inplace((char*)as+2)); Value nv=eval_expr(rt,needle), lv=eval_expr(rt,listname); int found=0,i; if(lv.type==V_LIST){ for(i=0;icount;i++) if(value_compare(nv,lv.list->items[i])==0){ found=i+1; break; } } rt_set(rt,var,v_num(found)); free(needle); free(listname); free(var); value_free(nv); value_free(lv); } return; } if(strcmp(up,"REMOVE")==0){ const char *from=find_word_ci(t,"FROM"); if(from){ char *needle=substr(t+6,from); char *listname=xstrdup(trim_inplace((char*)from+4)); Value nv=eval_expr(rt,needle), lv=rt_get(rt,listname); int i,j; if(lv.type==V_LIST){ for(i=0;icount;i++) if(value_compare(nv,lv.list->items[i])==0){ for(j=i;jcount-1;j++) lv.list->items[j]=lv.list->items[j+1]; lv.list->count--; break; } rt_set(rt,listname,lv); } free(needle); free(listname); } return; } if(strcmp(up,"TRY")==0){ int catchpos=-1,end=match_block(p,pc,"TRY","ENDTRY","CATCH",&catchpos); exec_range(rt,p,pc+1,catchpos>=0?catchpos:end); if(rt->error){ rt_clear_error(rt); if(catchpos>=0) exec_range(rt,p,catchpos+1,end); } *pcp=end<0?pc:end; return; } if(strcmp(up,"RAISE")==0){ char *e=expr_after_word(t,"RAISE"); Value v=eval_expr(rt,e); char *s=v_to_string(v); rt_error(rt,p->path,pc+1,"%s",s); free(s); free(e); return; } diff --git a/tools/validate_find_expression_cleanup.py b/tools/validate_find_expression_cleanup.py new file mode 100644 index 0000000..ed581fb --- /dev/null +++ b/tools/validate_find_expression_cleanup.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""Regression check for FIND expression temporaries.""" +from pathlib import Path +import os +import subprocess +import tempfile + +ROOT = Path(__file__).resolve().parent.parent + + +def main() -> int: + if os.name == "nt": + print("SKIP: LeakSanitizer check is POSIX-only") + return 0 + with tempfile.TemporaryDirectory(prefix="claro-find-cleanup-") as tmp: + tmp_path = Path(tmp) + binary = tmp_path / "claro-lsan" + script = tmp_path / "find_expressions.claro" + script.write_text( + "SET items TO LIST\n" + + 'FIND "Missing" IN items AS position\n' * 2000 + + "SAY position\n", + encoding="utf-8", + ) + build = subprocess.run( + ["gcc", "-std=c99", "-O0", "-g", "-fsanitize=address,undefined", + "src/claro.c", "-o", str(binary), "-lm"], + cwd=ROOT, text=True, capture_output=True, + ) + if build.returncode: + print(build.stdout, end="") + print(build.stderr, end="") + return build.returncode + run = subprocess.run( + [str(binary), str(script)], cwd=ROOT, text=True, capture_output=True, + env={**os.environ, "ASAN_OPTIONS": "detect_leaks=1"}, + ) + if run.returncode or "LeakSanitizer" in run.stderr or "SUMMARY:" in run.stderr: + print("FAIL: FIND expression values still leak") + print(run.stdout, end="") + print(run.stderr, end="") + return 1 + if run.stdout != "0\n": + print(f"FAIL: unexpected FIND probe output: {run.stdout!r}") + return 1 + print("PASS: FIND expression values are released") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())