From 6feabea6cb37edd62ef0c25d809d20b63e55bef6 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Mon, 17 Aug 2026 23:07:09 +0000 Subject: [PATCH] Add pure C89 Peppermint interpreter --- c89/Makefile | 10 + c89/README.md | 30 +++ c89/peppermint.c | 607 +++++++++++++++++++++++++++++++++++++++++++++++ c89/smoke.ppm | 4 + c89/test.sh | 27 +++ 5 files changed, 678 insertions(+) create mode 100644 c89/Makefile create mode 100644 c89/README.md create mode 100644 c89/peppermint.c create mode 100644 c89/smoke.ppm create mode 100755 c89/test.sh diff --git a/c89/Makefile b/c89/Makefile new file mode 100644 index 0000000..4d6fe81 --- /dev/null +++ b/c89/Makefile @@ -0,0 +1,10 @@ +CC ?= cc +CFLAGS ?= -std=c89 -pedantic -Wall -Wextra -Wconversion -Wshadow + +all: peppermint + +peppermint: peppermint.c + $(CC) $(CFLAGS) -o peppermint peppermint.c + +clean: + rm -f peppermint diff --git a/c89/README.md b/c89/README.md new file mode 100644 index 0000000..7190bdf --- /dev/null +++ b/c89/README.md @@ -0,0 +1,30 @@ +# Peppermint C89 port + +This directory is a clean native C89 implementation of the language behavior +present in the original TypeScript interpreter. + +Implemented subset: + +- `let name := expression` bindings +- integer and floating-point literals +- quoted strings with basic escapes +- `+`, `-`, `*`, `/`, and `%` +- string concatenation with `+` +- `#` comments and line-oriented programs +- `@argv` and `@argv:N` +- `@env:NAME` and `@exit` + +The original repository also contains a separate Brainfuck compiler/runtime; +that is not silently folded into this first language-core port. + +## Build + +Requires only a C compiler with C89 support: + +```sh +make +./peppermint smoke.ppm hello +``` + +The build uses `-std=c89 -pedantic` and has no TypeScript, Node.js, Beef, or +third-party runtime dependency. diff --git a/c89/peppermint.c b/c89/peppermint.c new file mode 100644 index 0000000..06ba4b9 --- /dev/null +++ b/c89/peppermint.c @@ -0,0 +1,607 @@ +/* + * Peppermint C89 interpreter. + * + * This implements the language behavior present in the original TypeScript + * front end: let bindings, integer/real literals, strings, arithmetic, + * comments, and @argv/@env/@exit builtins. + */ +#include +#include +#include +#include + +#define PM_NAME_MAX 128 +#define PM_TEXT_MAX 4096 +#define PM_MAX_VARS 256 + +typedef enum { + PM_NONE, + PM_INT, + PM_REAL, + PM_STRING +} PMType; + +typedef struct { + PMType type; + long integer; + double real; + char *text; +} PMValue; + +typedef struct { + char name[PM_NAME_MAX]; + PMValue value; +} PMVariable; + +typedef struct { + const char *source; + const char *filename; + long line; + int argc; + char **argv; + PMVariable vars[PM_MAX_VARS]; + int var_count; + int stopped; +} PMContext; + +static void pm_free_value(PMValue *value) +{ + if (value->type == PM_STRING && value->text != NULL) { + free(value->text); + } + value->type = PM_NONE; + value->text = NULL; + value->integer = 0L; + value->real = 0.0; +} + +static char *pm_copy_text(const char *start, size_t length) +{ + char *result; + result = (char *)malloc(length + 1U); + if (result == NULL) { + return NULL; + } + memcpy(result, start, length); + result[length] = '\0'; + return result; +} + +static void pm_error(PMContext *context, const char *message) +{ + fprintf(stderr, "%s:%ld: %s\n", context->filename, context->line, message); +} + +static void pm_skip_space(PMContext *context, const char **cursor) +{ + const char *p; + (void)context; + p = *cursor; + for (;;) { + while (*p == ' ' || *p == '\t' || *p == '\r') { + p++; + } + if (*p == '#') { + while (*p != '\0' && *p != '\n') { + p++; + } + } else { + break; + } + } + *cursor = p; +} + +static int pm_is_identifier_start(int c) +{ + return isalpha((unsigned char)c) || c == '_'; +} + +static int pm_is_identifier_char(int c) +{ + return isalnum((unsigned char)c) || c == '_'; +} + +static int pm_match(PMContext *context, const char **cursor, const char *word) +{ + const char *p; + size_t length; + (void)context; + p = *cursor; + length = strlen(word); + if (strncmp(p, word, length) == 0 && + !pm_is_identifier_char((unsigned char)p[length])) { + *cursor = p + length; + return 1; + } + return 0; +} + +static int pm_make_none(PMValue *value) +{ + value->type = PM_NONE; + value->integer = 0L; + value->real = 0.0; + value->text = NULL; + return 1; +} + +static int pm_make_integer(PMValue *value, long number) +{ + pm_make_none(value); + value->type = PM_INT; + value->integer = number; + return 1; +} + +static int pm_make_real(PMValue *value, double number) +{ + pm_make_none(value); + value->type = PM_REAL; + value->real = number; + return 1; +} + +static int pm_make_string(PMValue *value, const char *start, size_t length) +{ + pm_make_none(value); + value->text = pm_copy_text(start, length); + if (value->text == NULL) { + return 0; + } + value->type = PM_STRING; + return 1; +} + +static int pm_copy_value(PMValue *destination, const PMValue *source) +{ + pm_free_value(destination); + destination->type = source->type; + destination->integer = source->integer; + destination->real = source->real; + destination->text = NULL; + if (source->type == PM_STRING) { + if (source->text == NULL) { + return 1; + } + destination->text = pm_copy_text(source->text, strlen(source->text)); + if (destination->text == NULL) { + destination->type = PM_NONE; + return 0; + } + } + return 1; +} + +static PMVariable *pm_find_variable(PMContext *context, const char *name) +{ + int i; + for (i = 0; i < context->var_count; i++) { + if (strcmp(context->vars[i].name, name) == 0) { + return &context->vars[i]; + } + } + return NULL; +} + +static int pm_set_variable(PMContext *context, const char *name, + const PMValue *value) +{ + PMVariable *variable; + variable = pm_find_variable(context, name); + if (variable == NULL) { + if (context->var_count >= PM_MAX_VARS || strlen(name) >= PM_NAME_MAX) { + pm_error(context, "too many or oversized variables"); + return 0; + } + variable = &context->vars[context->var_count]; + strcpy(variable->name, name); + pm_make_none(&variable->value); + context->var_count++; + } + return pm_copy_value(&variable->value, value); +} + +static int pm_parse_expression(PMContext *context, const char **cursor, + PMValue *value); + +static int pm_parse_string(PMContext *context, const char **cursor, + PMValue *value) +{ + const char *p; + const char *start; + char buffer[PM_TEXT_MAX]; + size_t length; + p = *cursor; + p++; + start = p; + length = 0U; + while (*p != '\0' && *p != '"') { + if (*p == '\\' && p[1] != '\0') { + p++; + if (*p == 'n') { + if (length < PM_TEXT_MAX - 1U) buffer[length++] = '\n'; + } else if (*p == 'r') { + if (length < PM_TEXT_MAX - 1U) buffer[length++] = '\r'; + } else if (*p == 't') { + if (length < PM_TEXT_MAX - 1U) buffer[length++] = '\t'; + } else { + if (length < PM_TEXT_MAX - 1U) buffer[length++] = *p; + } + p++; + } else { + if (length < PM_TEXT_MAX - 1U) buffer[length++] = *p; + p++; + } + } + (void)start; + if (*p != '"') { + pm_error(context, "string not closed"); + return 0; + } + buffer[length] = '\0'; + if (!pm_make_string(value, buffer, length)) { + pm_error(context, "out of memory"); + return 0; + } + *cursor = p + 1; + return 1; +} + +static int pm_parse_primary(PMContext *context, const char **cursor, + PMValue *value) +{ + const char *p; + char name[PM_NAME_MAX]; + size_t length; + char *end; + long integer; + double real; + PMVariable *variable; + p = *cursor; + pm_skip_space(context, &p); + if (*p == '"') { + if (!pm_parse_string(context, &p, value)) return 0; + *cursor = p; + return 1; + } + if (*p == '(') { + p++; + if (!pm_parse_expression(context, &p, value)) return 0; + pm_skip_space(context, &p); + if (*p != ')') { + pm_error(context, "expected ')' "); + pm_free_value(value); + return 0; + } + *cursor = p + 1; + return 1; + } + if (isdigit((unsigned char)*p) || (*p == '.' && isdigit((unsigned char)p[1]))) { + real = strtod(p, &end); + if (end == p) { + pm_error(context, "invalid number"); + return 0; + } + if (strchr(p, '.') != NULL && strchr(p, '.') < end) { + pm_make_real(value, real); + } else { + integer = strtol(p, &end, 10); + pm_make_integer(value, integer); + } + *cursor = end; + return 1; + } + if (pm_is_identifier_start((unsigned char)*p)) { + length = 0U; + while (pm_is_identifier_char((unsigned char)p[length])) { + if (length < PM_NAME_MAX - 1U) name[length] = p[length]; + length++; + } + if (length >= PM_NAME_MAX) { + pm_error(context, "identifier too long"); + return 0; + } + name[length] = '\0'; + variable = pm_find_variable(context, name); + if (variable == NULL) { + pm_error(context, "unknown identifier"); + return 0; + } + if (!pm_copy_value(value, &variable->value)) { + pm_error(context, "out of memory"); + return 0; + } + *cursor = p + length; + return 1; + } + pm_error(context, "expected expression"); + return 0; +} + +static int pm_numeric(const PMValue *value, double *number) +{ + if (value->type == PM_INT) { + *number = (double)value->integer; + return 1; + } + if (value->type == PM_REAL) { + *number = value->real; + return 1; + } + return 0; +} + +static int pm_apply_operator(PMContext *context, PMValue *left, char operator, + PMValue *right) +{ + double a; + double b; + double result; + int integer_result; + char *text; + size_t left_length; + size_t right_length; + if (operator == '+' && left->type == PM_STRING && right->type == PM_STRING) { + left_length = strlen(left->text); + right_length = strlen(right->text); + text = (char *)malloc(left_length + right_length + 1U); + if (text == NULL) { + pm_error(context, "out of memory"); + return 0; + } + memcpy(text, left->text, left_length); + memcpy(text + left_length, right->text, right_length + 1U); + pm_free_value(left); + left->type = PM_STRING; + left->text = text; + return 1; + } + if (!pm_numeric(left, &a) || !pm_numeric(right, &b)) { + pm_error(context, "arithmetic requires numeric values"); + return 0; + } + if (operator == '+') result = a + b; + else if (operator == '-') result = a - b; + else if (operator == '*') result = a * b; + else if (operator == '/') { + if (b == 0.0) { + pm_error(context, "division by zero"); + return 0; + } + result = a / b; + } else if (operator == '%') { + long ia; + long ib; + if (left->type != PM_INT || right->type != PM_INT || b == 0.0) { + pm_error(context, "modulo requires nonzero integers"); + return 0; + } + ia = left->integer; + ib = right->integer; + pm_free_value(left); + pm_make_integer(left, ia % ib); + return 1; + } else { + pm_error(context, "unknown operator"); + return 0; + } + integer_result = left->type == PM_INT && right->type == PM_INT && operator != '/'; + pm_free_value(left); + if (integer_result) { + pm_make_integer(left, (long)result); + } else { + pm_make_real(left, result); + } + return 1; +} + +static int pm_parse_expression(PMContext *context, const char **cursor, + PMValue *value) +{ + const char *p; + PMValue right; + char operator; + p = *cursor; + pm_make_none(value); + if (!pm_parse_primary(context, &p, value)) return 0; + for (;;) { + pm_skip_space(context, &p); + operator = *p; + if (operator != '+' && operator != '-' && operator != '*' && + operator != '/' && operator != '%') break; + p++; + pm_make_none(&right); + if (!pm_parse_primary(context, &p, &right)) { + pm_free_value(value); + return 0; + } + if (!pm_apply_operator(context, value, operator, &right)) { + pm_free_value(value); + pm_free_value(&right); + return 0; + } + pm_free_value(&right); + } + *cursor = p; + return 1; +} + +static void pm_print_value(const PMValue *value) +{ + if (value->type == PM_STRING) printf("%s\n", value->text); + else if (value->type == PM_INT) printf("%ld\n", value->integer); + else if (value->type == PM_REAL) printf("%g\n", value->real); + else printf("None\n"); +} + +static void pm_builtin(PMContext *context, const char **cursor) +{ + const char *p; + const char *start; + char name[PM_NAME_MAX]; + char *end; + long index; + size_t length; + p = *cursor + 1; + start = p; + while (pm_is_identifier_char((unsigned char)*p)) p++; + length = (size_t)(p - start); + if (length >= PM_NAME_MAX) { + pm_error(context, "builtin name too long"); + return; + } + memcpy(name, start, length); + name[length] = '\0'; + if (strcmp(name, "exit") == 0) { + context->stopped = 1; + *cursor = p; + return; + } + if (strcmp(name, "argv") == 0) { + if (*p == ':') { + index = strtol(p + 1, &end, 10); + if (end == p + 1 || index < 0L || index >= context->argc) { + printf("None\n"); + } else { + printf("%s\n", context->argv[index]); + } + p = end; + } else { + int i; + for (i = 0; i < context->argc; i++) { + if (i != 0) putchar(' '); + fputs(context->argv[i], stdout); + } + putchar('\n'); + } + } else if (strcmp(name, "env") == 0) { + const char *key; + const char *result; + key = NULL; + if (*p == ':') { + p++; + start = p; + while (pm_is_identifier_char((unsigned char)*p)) p++; + length = (size_t)(p - start); + if (length >= PM_NAME_MAX) length = PM_NAME_MAX - 1U; + memcpy(name, start, length); + name[length] = '\0'; + key = name; + } + result = key == NULL ? NULL : getenv(key); + printf("%s\n", result == NULL || result[0] == '\0' ? "None" : result); + } else { + pm_error(context, "unknown builtin"); + } + *cursor = p; +} + +static int pm_run(PMContext *context) +{ + const char *p; + PMValue value; + char name[PM_NAME_MAX]; + size_t length; + p = context->source; + context->line = 1L; + while (*p != '\0' && !context->stopped) { + pm_skip_space(context, &p); + if (*p == '\n') { + context->line++; + p++; + continue; + } + if (*p == '\0') break; + if (*p == '@') { + pm_builtin(context, &p); + } else if (pm_match(context, &p, "let")) { + pm_skip_space(context, &p); + length = 0U; + while (pm_is_identifier_char((unsigned char)p[length])) { + if (length < PM_NAME_MAX - 1U) name[length] = p[length]; + length++; + } + if (length == 0U || length >= PM_NAME_MAX) { + pm_error(context, "expected variable name"); + return 0; + } + name[length] = '\0'; + p += length; + pm_skip_space(context, &p); + if (p[0] != ':' || p[1] != '=') { + pm_error(context, "expected ':='"); + return 0; + } + p += 2; + pm_make_none(&value); + if (!pm_parse_expression(context, &p, &value)) return 0; + if (!pm_set_variable(context, name, &value)) { + pm_free_value(&value); + return 0; + } + pm_free_value(&value); + } else { + pm_make_none(&value); + if (!pm_parse_expression(context, &p, &value)) return 0; + pm_print_value(&value); + pm_free_value(&value); + } + while (*p == ' ' || *p == '\t' || *p == '\r') p++; + if (*p == '\n') { + context->line++; + p++; + } else if (*p != '\0' && *p != '#') { + pm_error(context, "unexpected characters"); + return 0; + } + } + return 1; +} + +static char *pm_read_file(const char *filename) +{ + FILE *file; + long size; + char *data; + size_t count; + file = fopen(filename, "rb"); + if (file == NULL) return NULL; + if (fseek(file, 0L, SEEK_END) != 0) { fclose(file); return NULL; } + size = ftell(file); + if (size < 0L || fseek(file, 0L, SEEK_SET) != 0) { fclose(file); return NULL; } + data = (char *)malloc((size_t)size + 1U); + if (data == NULL) { fclose(file); return NULL; } + count = fread(data, 1U, (size_t)size, file); + fclose(file); + data[count] = '\0'; + return data; +} + +int main(int argc, char **argv) +{ + PMContext context; + char *source; + int i; + int result; + if (argc < 2) { + fprintf(stderr, "usage: peppermint [args...]\n"); + return 2; + } + source = pm_read_file(argv[1]); + if (source == NULL) { + fprintf(stderr, "peppermint: cannot read %s\n", argv[1]); + return 2; + } + context.source = source; + context.filename = argv[1]; + context.argc = argc - 2; + context.argv = argv + 2; + context.var_count = 0; + context.stopped = 0; + for (i = 0; i < PM_MAX_VARS; i++) pm_make_none(&context.vars[i].value); + result = pm_run(&context) ? 0 : 1; + for (i = 0; i < context.var_count; i++) pm_free_value(&context.vars[i].value); + free(source); + return result; +} diff --git a/c89/smoke.ppm b/c89/smoke.ppm new file mode 100644 index 0000000..f1c3a8f --- /dev/null +++ b/c89/smoke.ppm @@ -0,0 +1,4 @@ +let d := 9 +let total := d + 3 +"hello" + " peppermint" +@argv:0 diff --git a/c89/test.sh b/c89/test.sh new file mode 100755 index 0000000..eb3cd0b --- /dev/null +++ b/c89/test.sh @@ -0,0 +1,27 @@ +#!/bin/sh +set -eu + +make clean +make +output=$(./peppermint smoke.ppm argument) +expected='hello peppermint +argument' +if [ "$output" != "$expected" ]; then + printf '%s\n' "unexpected smoke output:" "$output" >&2 + exit 1 +fi + +cat > /tmp/peppermint-c89-test.ppm <<'EOF' +let a := 7 +let b := a * 3 +b + 2 +@env:PEPPERMINT_TEST_MISSING +EOF +output=$(PEPPERMINT_TEST_MISSING= ./peppermint /tmp/peppermint-c89-test.ppm) +expected='23 +None' +if [ "$output" != "$expected" ]; then + printf '%s\n' "unexpected arithmetic output:" "$output" >&2 + exit 1 +fi +printf '%s\n' 'C89 smoke tests passed'