diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..1ed0dc7 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,13 @@ +# Shell and helper scripts must use LF line endings on every platform. +# On Windows, a CRLF line ending on the shebang line (e.g. "#!/usr/bin/env +# bash\r") makes the interpreter lookup fail, so force LF regardless of the +# user's core.autocrlf setting. +*.sh text eol=lf +*.perl text eol=lf + +# The launcher and its real implementation are extension-less / .sh scripts. +ell text eol=lf +ell.sh text eol=lf + +# Templates and configuration examples are line-oriented text; keep them LF. +*.json text eol=lf diff --git a/README.md b/README.md index bb61036..241f2d1 100644 --- a/README.md +++ b/README.md @@ -20,32 +20,52 @@ A command-line interface for LLMs written in Bash. To use ell, you need the following: - bash-4.1 or later and coreutils / OS X utilities -- jq (For parsing JSON) - curl (For sending HTTPS requests) - perl (Not necessary if you don't use record mode. For PCRE. POSIX bash doesn't support look-ahead and look-behind regular expressions) - util-linux (Not necessary if you don't use record mode. For `script` command to record terminal input and output) ## Install -``` -git clone --depth 1 https://github.com/simonmysun/ell.git ~/.ellrc.d -echo 'export PATH="${HOME}/.ellrc.d:${PATH}"' >> ~/.bashrc +```bash +git clone --depth 1 https://github.com/simonmysun/ell.git \ + "${XDG_DATA_HOME:-$HOME/.local/share}/ell" +echo 'export PATH="${XDG_DATA_HOME:-$HOME/.local/share}/ell:$PATH"' >> ~/.bashrc ``` or +```bash +git clone --depth 1 git@github.com:simonmysun/ell.git \ + "${XDG_DATA_HOME:-$HOME/.local/share}/ell" +echo 'export PATH="${XDG_DATA_HOME:-$HOME/.local/share}/ell:$PATH"' >> ~/.bashrc ``` -git clone --depth 1 git@github.com:simonmysun/ell.git ~/.ellrc.d -echo 'export PATH="${HOME}/.ellrc.d:${PATH}"' >> ~/.bashrc -``` -This will clone the repository into `.ellrc.d` in your home directory and add it to your PATH. +This clones the repository into `${XDG_DATA_HOME:-$HOME/.local/share}/ell` +(following the [XDG Base Directory Specification](https://specifications.freedesktop.org/basedir-spec/basedir-spec-latest.html)) +and adds it to your `PATH`. You may clone it anywhere you like; only the +directory on your `PATH` matters. + +> **Upgrading from an older install?** The previous layout +> (`git clone ... ~/.ellrc.d` with configuration in `~/.ellrc`) still works: +> `~/.ellrc` is still read, and templates and plugins under `~/.ellrc.d` are +> still picked up. No migration is required. + +### Windows + +ell is a Bash program, so run it from a Bash environment such as **Git Bash**, +**MSYS2**, **Cygwin** or **WSL**. + +The `ell` command is a plain wrapper script (not a symbolic link), so it works +even when git does not create symlinks on checkout — the default on Windows +unless Developer Mode or administrator rights are available. No extra setup is +required; clone the repository and add its directory to your `PATH` as shown +above. You invoke it the same way, e.g. `ell "your prompt"`. ## Configuration See [Configuration](docs/Configuration.md). -Here's an example configuration to use `gemini-1.5-flash` from Google. You need to set these variables in your `~/.ellrc`: +Here's an example configuration to use `gemini-1.5-flash` from Google. You need to set these variables in your config file, `${XDG_CONFIG_HOME:-$HOME/.config}/ell/config` (the legacy `~/.ellrc` also still works): ```ini ELL_API_STYLE=gemini @@ -173,6 +193,23 @@ See [Risks Consideration](docs/Risk_Consideration.md). - https://github.com/sigoden/aichat A CLI tool talks to various LLM providers, written in Rust. - https://github.com/npiv/chatblade A CLI Swiss Army Knife for ChatGPT, written in python +## Testing + +The test suite is meant to be run inside Docker. The tests hard-code the +container path `/ell` (the repository is mounted there read-only) and rely on +a clean, predictable environment, so running them directly on your host is not +supported and will fail (e.g. templates are looked up under `/ell/templates/`). + +Run the full suite against the supported Bash versions with: + +```bash +bash tests/docker.sh +``` + +This mounts the repository into `bash:4.1` and `bash:5.2` containers and runs +`tests/entry.sh`, which installs the runtime dependencies and executes every +test (`logging`, `piping`, `templating`, `parse_output` and `redaction`). + ## Contributing Contributions are welcome! If you have any ideas, suggestions, or bug reports, please open an issue or submit a pull request. diff --git a/docs/Configuration.md b/docs/Configuration.md index 4547d53..b07ef6c 100644 --- a/docs/Configuration.md +++ b/docs/Configuration.md @@ -8,9 +8,11 @@ ell can be configured in three ways (in order of precedence, from lowest to high - environment variables - command line arguments -The configuration files are read and applied in the following order: +The configuration files are read and applied in the following order (later +files override earlier ones): -- `~/.ellrc` +- `${XDG_CONFIG_HOME:-$HOME/.config}/ell/config` (the main config, following the [XDG Base Directory Specification](https://specifications.freedesktop.org/basedir-spec/basedir-spec-latest.html)) +- `~/.ellrc` (legacy location, still read for backward compatibility) - `.ellrc` in the current directory - `$ELL_CONFIG` specified in the environment variables or command line arguments. @@ -23,11 +25,11 @@ If you are running ell in a relatively hostile environment, it is recommended to The following variables can be set in the configuration files, environment variables: - `ELL_LOG_LEVEL`: The log level of the logger. The default is `2`. A log level of `0` will log everything. A log level of `3` will log token usage. -- `ELL_CONFIG`: The configuration file to use. The default is `~/.ellrc`. +- `ELL_CONFIG`: An extra configuration file to load, applied last (highest precedence among files). Unset by default. The standard config files (`${XDG_CONFIG_HOME:-$HOME/.config}/ell/config`, `~/.ellrc`, `./.ellrc`) are always read regardless of this variable. - `ELL_LLM_MODEL`: The model to use. Default is `gpt-4o-mini`. - `ELL_LLM_TEMPERATURE`: The temperature of the model. The default is `0.6`. - `ELL_LLM_MAX_TOKENS`: The maximum number of tokens to generate. The default is `4096`. -- `ELL_TEMPLATE_PATH`: The path to the templates. The default is `~/.ellrc.d/templates`. +- `ELL_TEMPLATE_PATH`: Force templates to be loaded from this single directory (note the trailing slash, e.g. `/path/to/templates/`). When unset (the default), templates are searched, in order, under `${XDG_CONFIG_HOME:-$HOME/.config}/ell/templates/`, `${XDG_DATA_HOME:-$HOME/.local/share}/ell/templates/`, `~/.ellrc.d/templates/` (legacy) and the `templates/` directory bundled with ell. This lets you override a bundled template by placing a file with the same name under your XDG config directory. - `ELL_TEMPLATE`: The template to use. The default is `default`. The file extension is not needed. - `ELL_INPUT_FILE`: The input file to use. If specified, it will override the prompt given in command line arguments. Setting this to `-` will let ell always read from stdin. - `ELL_RECORD`: This is used for controlling whether record mode is on. It should be set to `false` unless you want to disable recording. diff --git a/docs/Plugins.md b/docs/Plugins.md index 77b97f6..d15bd4b 100644 --- a/docs/Plugins.md +++ b/docs/Plugins.md @@ -24,11 +24,22 @@ Ell supports plugins to extend its functionality through a hook system. Currentl - `post_llm`: Called after the response is received and decoded from the language model. - `pre_output`: Called before the output is sent to the user. -Plugins should be placed in the `./plugins` directory related to the ell script, typically located at `~/.ellrc.d/plugins` if you follow the installation instructions in the readme. +Plugins are discovered from the `plugins/` directory under each of the +following roots, searched in this priority order: -Each plugin should be a folder containing executable shell scripts. The file name should follow the format `XX_${HOOK_NAME}.sh`, where `XX` is a number that determines the execution order among other plugins. For example, the paginator plugin is placed in `~/.ellrc.d/plugins/paginator/90_pre_output.sh`. +- `${XDG_CONFIG_HOME:-$HOME/.config}/ell/plugins/` (your own plugins) +- `${XDG_DATA_HOME:-$HOME/.local/share}/ell/plugins/` +- `~/.ellrc.d/plugins/` (legacy location, still supported) +- the `plugins/` directory bundled with ell (built-in plugins) -Plugin scripts are executed in ascending numerical order and piped to each other. +If the same plugin hook (identified by `/`) exists in +more than one root, only the highest-priority copy runs, so you can drop a +plugin with the same name under your XDG config directory to override a +built-in one. + +Each plugin should be a folder containing executable shell scripts. The file name should follow the format `XX_${HOOK_NAME}.sh`, where `XX` is a number that determines the execution order among other plugins. For example, the built-in paginator plugin is placed in `plugins/paginator/90_pre_output.sh` (so a user copy would live at `${XDG_CONFIG_HOME:-$HOME/.config}/ell/plugins/paginator/90_pre_output.sh`). + +Plugin scripts are executed in ascending numerical order (across all roots, ordered by `/`) and piped to each other. It is recommended to write plugins in a streaming manner. diff --git a/ell b/ell deleted file mode 120000 index fe3f946..0000000 --- a/ell +++ /dev/null @@ -1 +0,0 @@ -ell.sh \ No newline at end of file diff --git a/ell b/ell new file mode 100755 index 0000000..64fd4a0 --- /dev/null +++ b/ell @@ -0,0 +1,28 @@ +#!/usr/bin/env bash + +# Thin launcher for ell. +# +# This file used to be a symbolic link to ell.sh, but git symlinks are not +# checked out as links on Windows (unless core.symlinks is enabled, which +# usually requires Developer Mode or administrator rights). There they become +# plain text files containing the link target, which cannot be executed. +# +# A real wrapper script works identically on Linux, macOS and Windows (Git +# Bash / MSYS2 / Cygwin / WSL), so `ell` stays a valid, extension-less command +# everywhere. It simply resolves its own directory and execs the real script. + +# Resolve the directory this launcher lives in, following any symlinks that +# may still exist (e.g. a link created in ~/.local/bin on Unix). +SOURCE="${BASH_SOURCE[0]:-${0}}"; +while [ -h "${SOURCE}" ]; do + DIR="$(cd -P "$(dirname "${SOURCE}")" >/dev/null 2>&1 && pwd)"; + SOURCE="$(readlink "${SOURCE}")"; + # If the link target is relative, resolve it against the link's directory. + case "${SOURCE}" in + /*) ;; + *) SOURCE="${DIR}/${SOURCE}"; ;; + esac +done +ELL_DIR="$(cd -P "$(dirname "${SOURCE}")" >/dev/null 2>&1 && pwd)"; + +exec "${ELL_DIR}/ell.sh" "${@}"; diff --git a/ell.sh b/ell.sh index 478e805..0c3c69c 100755 --- a/ell.sh +++ b/ell.sh @@ -22,6 +22,8 @@ BASE_DIR=$(dirname "${0}"); . "${BASE_DIR}/helpers/parse_arguments.sh"; . "${BASE_DIR}/helpers/load_config.sh"; . "${BASE_DIR}/helpers/piping.sh"; +. "${BASE_DIR}/helpers/json.sh"; +. "${BASE_DIR}/helpers/resolve_paths.sh"; logging_debug "Starting ${0}"; @@ -37,7 +39,9 @@ load_config; : "${ELL_LLM_MODEL:=gpt-4o-mini}"; : "${ELL_LLM_TEMPERATURE:=0.6}"; : "${ELL_LLM_MAX_TOKENS:=4096}"; -: "${ELL_TEMPLATE_PATH:="${HOME}/.ellrc.d/templates/"}"; +# ELL_TEMPLATE_PATH is intentionally left unset by default: templates are +# resolved through resolve_template() across the XDG and bundled search roots. +# Setting it explicitly (e.g. via -T) forces that single directory instead. : "${ELL_TEMPLATE:=default-openai}"; : "${ELL_INPUT_FILE:=""}"; : "${ELL_RECORD:="false"}"; @@ -83,9 +87,9 @@ export COLUMNS; # Logging_debug "Decorating the generate_completion to apply hooks before and after"; eval "$(printf "orig_"; command -V generate_completion | tail -n +2)"; generate_completion() { - pre_llm_hooks=$(ls ${BASE_DIR}/plugins/*/*_pre_llm.sh 2>/dev/null | sort -k3 -t/); + pre_llm_hooks=$(list_plugin_hooks _pre_llm.sh); logging_debug "Pre LLM hooks: ${pre_llm_hooks}"; - post_llm_hooks=$(ls ${BASE_DIR}/plugins/*/*_post_llm.sh 2>/dev/null | sort -k3 -t/); + post_llm_hooks=$(list_plugin_hooks _post_llm.sh); logging_debug "Post LLM hooks: ${post_llm_hooks}"; piping "${pre_llm_hooks[@]}" \ | orig_generate_completion \ @@ -116,11 +120,17 @@ if [ "x${ELL_RECORD}" = "xtrue" ] || [ "x${ELL_INTERACTIVE}" = "xtrue" ] && [ "x exit 0; fi -# Logging_debug "Checking if the template is available"; -if [ ! -f "${ELL_TEMPLATE_PATH}${ELL_TEMPLATE}.json" ]; then - logging_fatal "Template not found: ${ELL_TEMPLATE_PATH}${ELL_TEMPLATE}.json"; +# Logging_debug "Resolving the template across the search roots"; +ELL_TEMPLATE_FILE="$(resolve_template "${ELL_TEMPLATE}")"; +if [ -z "${ELL_TEMPLATE_FILE}" ]; then + if [ -n "${ELL_TEMPLATE_PATH}" ]; then + logging_fatal "Template not found: ${ELL_TEMPLATE_PATH}${ELL_TEMPLATE}.json"; + else + logging_fatal "Template not found: ${ELL_TEMPLATE}.json (searched XDG config/data, ~/.ellrc.d and ${BASE_DIR})"; + fi exit 1; fi +logging_debug "Using template: ${ELL_TEMPLATE_FILE}"; # Logging_debug "Checking if we are going to read from a file"; if [ -n "${ELL_INPUT_FILE}" ]; then @@ -142,9 +152,9 @@ else fi # Logging_debug "Loading the post_input and pre_output hooks"; -post_input_hooks=$(ls ${BASE_DIR}/plugins/*/*_post_input.sh 2>/dev/null | sort -k3 -t/); +post_input_hooks=$(list_plugin_hooks _post_input.sh); logging_debug "Post input hooks: ${post_input_hooks}"; -pre_output_hooks=$(ls ${BASE_DIR}/plugins/*/*_pre_output.sh 2>/dev/null | sort -k3 -t/); +pre_output_hooks=$(list_plugin_hooks _pre_output.sh); logging_debug "Pre output hooks: ${pre_output_hooks}"; # Logging_debug "Checking if we are going to enter interactive mode"; @@ -161,7 +171,7 @@ if [ "x${ELL_INTERACTIVE}" = "xtrue" ]; then export SHELL_CONTEXT="$(tail -c 3000 "${ELL_TMP_SHELL_LOG}" | "${BASE_DIR}/helpers/render_to_text.perl" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g'| awk '{printf "%s\\n", $0}')"; fi PAYLOAD="$(eval "cat < +# Parse a document once. Results are stored in the associative array +# `JSON` keyed by path. Returns non-zero if the text is not valid JSON. +# +# json_get +# Print the decoded value stored at from the last json_parse. +# Returns non-zero if the path is absent. +# +# json_has +# Return zero (success) if exists and is not JSON null. +# +# json_is_valid +# Return zero if is a complete, valid JSON document. +# +# The parser stores fully decoded string values (escape sequences and +# \uXXXX sequences are turned into their UTF-8 bytes), so callers get the +# same output that `jq -r` produced previously. + +# Internal parser state. +declare -A JSON=(); # path -> decoded scalar value +declare -A JSON_TYPE=(); # path -> type: object|array|string|number|bool|null +_JSON_S=""; # the input string being parsed +_JSON_I=0; # current index into _JSON_S +_JSON_N=0; # length of _JSON_S + +# _json_error: record failure position for debugging and return non-zero. +_json_error() { + _JSON_ERR="${1} at offset ${_JSON_I}"; + return 1; +} + +# _json_skip_ws: advance the cursor past insignificant whitespace. +_json_skip_ws() { + local c; + while [ "${_JSON_I}" -lt "${_JSON_N}" ]; do + c="${_JSON_S:${_JSON_I}:1}"; + case "${c}" in + ' '|$'\t'|$'\n'|$'\r') _JSON_I=$((_JSON_I + 1)); ;; + *) break; ;; + esac + done +} + +# _json_parse_string: parse a JSON string starting at the opening quote and +# leave the fully decoded value in _JSON_STR. Advances the cursor past the +# closing quote. +_json_parse_string() { + local out="" c hex code + # Skip the opening quote. + _JSON_I=$((_JSON_I + 1)); + while [ "${_JSON_I}" -lt "${_JSON_N}" ]; do + c="${_JSON_S:${_JSON_I}:1}"; + if [ "${c}" = '"' ]; then + _JSON_I=$((_JSON_I + 1)); + _JSON_STR="${out}"; + return 0; + elif [ "${c}" = '\' ]; then + _JSON_I=$((_JSON_I + 1)); + c="${_JSON_S:${_JSON_I}:1}"; + case "${c}" in + '"') out="${out}\""; _JSON_I=$((_JSON_I + 1)); ;; + '\') out="${out}\\"; _JSON_I=$((_JSON_I + 1)); ;; + '/') out="${out}/"; _JSON_I=$((_JSON_I + 1)); ;; + b) out="${out}"$'\b'; _JSON_I=$((_JSON_I + 1)); ;; + f) out="${out}"$'\f'; _JSON_I=$((_JSON_I + 1)); ;; + n) out="${out}"$'\n'; _JSON_I=$((_JSON_I + 1)); ;; + r) out="${out}"$'\r'; _JSON_I=$((_JSON_I + 1)); ;; + t) out="${out}"$'\t'; _JSON_I=$((_JSON_I + 1)); ;; + u) + hex="${_JSON_S:$((_JSON_I + 1)):4}"; + if [ "${#hex}" -ne 4 ]; then + _json_error "truncated \\u escape"; + return 1; + fi + code=$((16#${hex})); + _JSON_I=$((_JSON_I + 5)); + # Handle UTF-16 surrogate pairs. + if [ "${code}" -ge 55296 ] && [ "${code}" -le 56319 ]; then + if [ "${_JSON_S:${_JSON_I}:2}" = '\u' ]; then + local hex2 lo + hex2="${_JSON_S:$((_JSON_I + 2)):4}"; + lo=$((16#${hex2})); + _JSON_I=$((_JSON_I + 6)); + code=$(( (code - 55296) * 1024 + (lo - 56320) + 65536 )); + fi + fi + # Convert the code point to UTF-8 bytes and append them to out. + # Using printf -v (rather than command substitution) preserves + # bytes that would otherwise be stripped, such as \u000a (newline). + _json_codepoint_to_utf8 "${code}"; + out="${out}${_JSON_UTF8}"; + ;; + *) + _json_error "invalid escape \\${c}"; + return 1; + ;; + esac + else + out="${out}${c}"; + _JSON_I=$((_JSON_I + 1)); + fi + done + _json_error "unterminated string"; + return 1; +} + +# _json_codepoint_to_utf8: encode a Unicode code point as UTF-8 and leave the +# resulting bytes in _JSON_UTF8. The result is built with printf -v so that +# no bytes (including newline / carriage return) are lost. +_json_codepoint_to_utf8() { + local code="${1}" esc; + _JSON_UTF8=""; + if [ "${code}" -eq 0 ]; then + # A literal NUL cannot be represented in a bash string; emit nothing. + return 0; + fi + # Build a string of \xHH escapes, then let printf decode it into raw bytes. + # printf -v keeps every byte, including newline (0x0a) and CR (0x0d) that + # command substitution would strip. + if [ "${code}" -lt 128 ]; then + printf -v esc '\\x%02x' "${code}"; + elif [ "${code}" -lt 2048 ]; then + printf -v esc '\\x%02x\\x%02x' \ + $(( (code >> 6) | 192 )) \ + $(( (code & 63) | 128 )); + elif [ "${code}" -lt 65536 ]; then + printf -v esc '\\x%02x\\x%02x\\x%02x' \ + $(( (code >> 12) | 224 )) \ + $(( ((code >> 6) & 63) | 128 )) \ + $(( (code & 63) | 128 )); + else + printf -v esc '\\x%02x\\x%02x\\x%02x\\x%02x' \ + $(( (code >> 18) | 240 )) \ + $(( ((code >> 12) & 63) | 128 )) \ + $(( ((code >> 6) & 63) | 128 )) \ + $(( (code & 63) | 128 )); + fi + printf -v _JSON_UTF8 '%b' "${esc}"; +} + +# _json_parse_literal: parse true, false or null starting at the cursor. +_json_parse_literal() { + local path="${1}"; + if [ "${_JSON_S:${_JSON_I}:4}" = "true" ]; then + JSON["${path}"]="true"; JSON_TYPE["${path}"]="bool"; + _JSON_I=$((_JSON_I + 4)); + return 0; + elif [ "${_JSON_S:${_JSON_I}:5}" = "false" ]; then + JSON["${path}"]="false"; JSON_TYPE["${path}"]="bool"; + _JSON_I=$((_JSON_I + 5)); + return 0; + elif [ "${_JSON_S:${_JSON_I}:4}" = "null" ]; then + JSON["${path}"]=""; JSON_TYPE["${path}"]="null"; + _JSON_I=$((_JSON_I + 4)); + return 0; + fi + _json_error "invalid literal"; + return 1; +} + +# _json_parse_number: parse a number token starting at the cursor. +_json_parse_number() { + local path="${1}"; + local start="${_JSON_I}" c; + while [ "${_JSON_I}" -lt "${_JSON_N}" ]; do + c="${_JSON_S:${_JSON_I}:1}"; + case "${c}" in + [0-9]|-|+|.|e|E) _JSON_I=$((_JSON_I + 1)); ;; + *) break; ;; + esac + done + JSON["${path}"]="${_JSON_S:${start}:$((_JSON_I - start))}"; + JSON_TYPE["${path}"]="number"; + return 0; +} + +# _JSON_ROOT: sentinel key used for the document root value. bash associative +# arrays cannot use an empty subscript, so the empty path maps to this key. +_JSON_ROOT=$'\001root'; + +# _json_key: translate a logical path ("" for root) into a storage key. +_json_key() { + if [ -z "${1}" ]; then + printf '%s' "${_JSON_ROOT}"; + else + printf '%s' "${1}"; + fi +} + +# _json_parse_value: dispatch on the next token and parse a value into . +_json_parse_value() { + local path key c; + path="${1}"; + key="$(_json_key "${path}")"; + _json_skip_ws; + if [ "${_JSON_I}" -ge "${_JSON_N}" ]; then + _json_error "unexpected end of input"; + return 1; + fi + c="${_JSON_S:${_JSON_I}:1}"; + case "${c}" in + '{') _json_parse_object "${path}" "${key}"; return "${?}"; ;; + '[') _json_parse_array "${path}" "${key}"; return "${?}"; ;; + '"') + if ! _json_parse_string; then return 1; fi + JSON["${key}"]="${_JSON_STR}"; JSON_TYPE["${key}"]="string"; + return 0; + ;; + t|f|n) _json_parse_literal "${key}"; return "${?}"; ;; + -|[0-9]) _json_parse_number "${key}"; return "${?}"; ;; + *) _json_error "unexpected character '${c}'"; return 1; ;; + esac +} + +# _json_parse_object: parse an object into (storage key ) and its +# members. +_json_parse_object() { + local path="${1}" key="${2}" child mkey c; + JSON["${key}"]="[object]"; JSON_TYPE["${key}"]="object"; + _JSON_I=$((_JSON_I + 1)); # skip '{' + _json_skip_ws; + if [ "${_JSON_S:${_JSON_I}:1}" = '}' ]; then + _JSON_I=$((_JSON_I + 1)); + return 0; + fi + while true; do + _json_skip_ws; + if [ "${_JSON_S:${_JSON_I}:1}" != '"' ]; then + _json_error "expected object key"; + return 1; + fi + if ! _json_parse_string; then return 1; fi + mkey="${_JSON_STR}"; + _json_skip_ws; + if [ "${_JSON_S:${_JSON_I}:1}" != ':' ]; then + _json_error "expected ':'"; + return 1; + fi + _JSON_I=$((_JSON_I + 1)); + if [ -z "${path}" ]; then + child="${mkey}"; + else + child="${path}.${mkey}"; + fi + if ! _json_parse_value "${child}"; then return 1; fi + _json_skip_ws; + c="${_JSON_S:${_JSON_I}:1}"; + if [ "${c}" = ',' ]; then + _JSON_I=$((_JSON_I + 1)); + continue; + elif [ "${c}" = '}' ]; then + _JSON_I=$((_JSON_I + 1)); + return 0; + else + _json_error "expected ',' or '}'"; + return 1; + fi + done +} + +# _json_parse_array: parse an array into (storage key ) and its +# indexed elements. The element count is stored under ".length". +_json_parse_array() { + local path="${1}" key="${2}" idx=0 child c; + JSON["${key}"]="[array]"; JSON_TYPE["${key}"]="array"; + _JSON_I=$((_JSON_I + 1)); # skip '[' + _json_skip_ws; + if [ "${_JSON_S:${_JSON_I}:1}" = ']' ]; then + _JSON_I=$((_JSON_I + 1)); + JSON["${key}.length"]="0"; + return 0; + fi + while true; do + if [ -z "${path}" ]; then + child="${idx}"; + else + child="${path}.${idx}"; + fi + if ! _json_parse_value "${child}"; then return 1; fi + idx=$((idx + 1)); + _json_skip_ws; + c="${_JSON_S:${_JSON_I}:1}"; + if [ "${c}" = ',' ]; then + _JSON_I=$((_JSON_I + 1)); + continue; + elif [ "${c}" = ']' ]; then + _JSON_I=$((_JSON_I + 1)); + JSON["${key}.length"]="${idx}"; + return 0; + else + _json_error "expected ',' or ']'"; + return 1; + fi + done +} + +# json_parse: entry point. Parse and populate JSON / JSON_TYPE. +json_parse() { + JSON=(); + JSON_TYPE=(); + _JSON_S="${1}"; + _JSON_I=0; + _JSON_N="${#_JSON_S}"; + _JSON_ERR=""; + if ! _json_parse_value ""; then + return 1; + fi + _json_skip_ws; + # Trailing content after a complete value means the document is malformed. + if [ "${_JSON_I}" -lt "${_JSON_N}" ]; then + _json_error "trailing content"; + return 1; + fi + return 0; +} + +# json_get: print the value at , or return non-zero if missing. +json_get() { + local key; + key="$(_json_key "${1}")"; + if [ -z "${JSON_TYPE[${key}]+x}" ]; then + return 1; + fi + printf '%s' "${JSON[${key}]}"; + return 0; +} + +# json_has: succeed if exists and is not null. +json_has() { + local key; + key="$(_json_key "${1}")"; + if [ -z "${JSON_TYPE[${key}]+x}" ]; then + return 1; + fi + if [ "${JSON_TYPE[${key}]}" = "null" ]; then + return 1; + fi + return 0; +} + +# json_is_valid: succeed if is a complete, valid JSON document. +json_is_valid() { + json_parse "${1}" >/dev/null 2>&1; +} + +export -f json_parse json_get json_has json_is_valid 2>/dev/null; +export -f _json_parse_value _json_parse_object _json_parse_array 2>/dev/null; +export -f _json_parse_string _json_parse_number _json_parse_literal 2>/dev/null; +export -f _json_skip_ws _json_codepoint_to_utf8 _json_error _json_key 2>/dev/null; diff --git a/helpers/load_config.sh b/helpers/load_config.sh index 3bf52d4..4efa330 100644 --- a/helpers/load_config.sh +++ b/helpers/load_config.sh @@ -2,14 +2,26 @@ # Configuration loader. # Will not overwrite existing variables. -# Will read from $HOME/.ellrc, $PWD/.ellrc, and $ELL_CONFIG +# Sources configuration files in the following order (later files override +# earlier ones, but never override variables already set in the environment): +# 1. ${XDG_CONFIG_HOME:-$HOME/.config}/ell/config (XDG main config) +# 2. $HOME/.ellrc (legacy, kept for compat) +# 3. $PWD/.ellrc (per-project config) +# 4. $ELL_CONFIG (explicit override) load_config() { logging_debug "Storing current environment"; current_env=$(declare -p -x | sed -e 's/declare -x /export /'); set -o allexport; + + ELL_XDG_CONFIG="${XDG_CONFIG_HOME:-${HOME}/.config}/ell/config"; + if [ -f "${ELL_XDG_CONFIG}" ]; then + logging_debug "Loading config from ${ELL_XDG_CONFIG} (XDG)"; + . "${ELL_XDG_CONFIG}" + fi + if [ -f "${HOME}/.ellrc" ]; then - logging_debug "Loading config from ${HOME}/.ellrc (from \$HOME)"; + logging_debug "Loading config from ${HOME}/.ellrc (from \$HOME, legacy)"; . "${HOME}/.ellrc" fi diff --git a/helpers/resolve_paths.sh b/helpers/resolve_paths.sh new file mode 100644 index 0000000..93448f2 --- /dev/null +++ b/helpers/resolve_paths.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash + +# Path resolution helpers implementing the XDG Base Directory Specification +# while remaining backward compatible with the legacy ~/.ellrc.d layout. +# +# Templates and plugins are searched across several roots, in priority order: +# 1. ${XDG_CONFIG_HOME:-$HOME/.config}/ell (user overrides) +# 2. ${XDG_DATA_HOME:-$HOME/.local/share}/ell +# 3. $HOME/.ellrc.d (legacy install location) +# 4. ${BASE_DIR} (files shipped with ell) +# +# BASE_DIR is expected to be set by the caller (ell.sh) to the directory the +# ell script lives in, so the bundled templates and plugins are always found. + +# _ell_search_roots: print the ordered list of root directories to search, +# one per line. BASE_DIR is printed last so bundled resources act as the +# built-in fallback. +_ell_search_roots() { + printf '%s\n' "${XDG_CONFIG_HOME:-${HOME}/.config}/ell"; + printf '%s\n' "${XDG_DATA_HOME:-${HOME}/.local/share}/ell"; + printf '%s\n' "${HOME}/.ellrc.d"; + if [ -n "${BASE_DIR}" ]; then + printf '%s\n' "${BASE_DIR}"; + fi +} + +# resolve_template : print the full path to the first matching +# ".json" template found in the search roots. If ELL_TEMPLATE_PATH is +# set explicitly (e.g. via -T / --template-path) it is treated as a single +# directory and takes precedence, preserving the previous behaviour. +# Returns non-zero if no template file is found. +resolve_template() { + local name="${1}" root candidate; + if [ -n "${ELL_TEMPLATE_PATH}" ]; then + candidate="${ELL_TEMPLATE_PATH}${name}.json"; + if [ -f "${candidate}" ]; then + printf '%s' "${candidate}"; + return 0; + fi + return 1; + fi + while IFS= read -r root; do + candidate="${root}/templates/${name}.json"; + if [ -f "${candidate}" ]; then + printf '%s' "${candidate}"; + return 0; + fi + done < <(_ell_search_roots) + return 1; +} + +# list_plugin_hooks : print, one per line, every plugin hook +# script matching "*/" across all search roots' plugins/ +# directories. +# +# The search roots are visited in priority order, so if the same plugin +# (identified by "/") exists in more than one root, +# only the highest-priority copy is kept. This prevents a plugin bundled with +# ell from also running from a legacy ~/.ellrc.d clone. +# +# The surviving hooks are ordered by "/" so the numeric +# ordering prefix (e.g. 90_pre_output.sh) is respected regardless of which +# root a plugin lives in. +list_plugin_hooks() { + local suffix="${1}" root; + { + while IFS= read -r root; do + ls "${root}"/plugins/*/*"${suffix}" 2>/dev/null; + done < <(_ell_search_roots) + } | awk -F/ ' + { + key = $(NF-1) "/" $NF; + if (!(key in seen)) { + seen[key] = 1; + print key "\t" $0; + } + } + ' | sort | cut -f2-; +} + +export -f _ell_search_roots resolve_template list_plugin_hooks 2>/dev/null; diff --git a/llm_backends/gemini/generate_completion.sh b/llm_backends/gemini/generate_completion.sh index 4ece84a..5959bf2 100644 --- a/llm_backends/gemini/generate_completion.sh +++ b/llm_backends/gemini/generate_completion.sh @@ -14,17 +14,21 @@ generate_completion() { logging_debug "Response: ${response}"; exit 1; else + if ! json_parse "${response}"; then + logging_error "Unexpected format: ${response}"; + return; + fi # check if finishReason is present - if (echo "${response}" | jq -e '.candidates[0].finishReason' > /dev/null); then - if [ "x$(echo "${response}" | jq -r '.candidates[0].finishReason')" != "xSTOP" ]; then - logging_error "Unexpected finish reason: $(echo "${response}" | jq -r '.choices[0].finish_reason')"; + if json_has "candidates.0.finishReason"; then + if [ "x$(json_get "candidates.0.finishReason")" != "xSTOP" ]; then + logging_error "Unexpected finish reason: $(json_get "candidates.0.finishReason")"; else - echo "${response}" | jq -j -r '.candidates[0].content.parts[0].text'; + json_get "candidates.0.content.parts.0.text"; echo ""; - if (echo "${response}" | jq -e -r '.usageMetadata' > /dev/null); then - prompt_tokens=$(echo "${response}" | jq -j -r '.usageMetadata.promptTokenCount'); - completion_tokens=$(echo "${response}" | jq -j -r '.usageMetadata.candidatesTokenCount'); - total_tokens=$(echo "${response}" | jq -j -r '.usageMetadata.totalTokenCount'); + if json_has "usageMetadata"; then + prompt_tokens=$(json_get "usageMetadata.promptTokenCount"); + completion_tokens=$(json_get "usageMetadata.candidatesTokenCount"); + total_tokens=$(json_get "usageMetadata.totalTokenCount"); echo ''; logging_info "usage: prompt_tokens=${prompt_tokens}, completion_tokens=${completion_tokens}, total_tokens=${total_tokens}"; fi @@ -60,25 +64,22 @@ generate_completion() { else BUFFER="${BUFFER}${line}"; # trying to parse the buffer as JSON - jq -e . >/dev/null 2>&1 < /dev/null); then - echo "${BUFFER}" | jq -j -r '.candidates[0].content.parts[0].text'; + if json_parse "${BUFFER}"; then + if json_has "candidates.0.content.parts.0.text"; then + json_get "candidates.0.content.parts.0.text"; fi - if (echo "${BUFFER}" | jq -e -r '.candidates[0].finishReason' > /dev/null); then - stop_reason=$(echo "${BUFFER}" | jq -j -r '.candidates[0].finishReason'); + if json_has "candidates.0.finishReason"; then + stop_reason=$(json_get "candidates.0.finishReason"); if [ "x${stop_reason}" != "xSTOP" ]; then logging_error "Unexpected stop reason: ${stop_reason}"; break; fi fi # check if usageMetadata is present, gemini API v1beta sends usageMetadata in every chunk - if (echo "${BUFFER}" | jq -e -r '.usageMetadata' > /dev/null); then - prompt_tokens=$(echo "${BUFFER}" | jq -j -r '.usageMetadata.promptTokenCount'); - completion_tokens=$(echo "${BUFFER}" | jq -j -r '.usageMetadata.candidatesTokenCount'); - total_tokens=$(echo "${BUFFER}" | jq -j -r '.usageMetadata.totalTokenCount'); + if json_has "usageMetadata"; then + prompt_tokens=$(json_get "usageMetadata.promptTokenCount"); + completion_tokens=$(json_get "usageMetadata.candidatesTokenCount"); + total_tokens=$(json_get "usageMetadata.totalTokenCount"); fi PART_FINISHED=true; BUFFER=""; @@ -96,4 +97,4 @@ EOF fi } -export generate_completion; \ No newline at end of file +export generate_completion; diff --git a/llm_backends/openai/generate_completion.sh b/llm_backends/openai/generate_completion.sh index 7e38927..68922b6 100644 --- a/llm_backends/openai/generate_completion.sh +++ b/llm_backends/openai/generate_completion.sh @@ -14,17 +14,21 @@ generate_completion() { logging_debug "Response: ${response}"; exit 1; else + if ! json_parse "${response}"; then + logging_error "Unexpected format: ${response}"; + return; + fi # check if finish_reason is present - if (echo "${response}" | jq -e '.choices[0].finish_reason' > /dev/null); then - if [ "x$(echo "${response}" | jq -r '.choices[0].finish_reason')" != "xstop" ]; then - logging_error "Unexpected finish reason: $(echo "${response}" | jq -r '.choices[0].finish_reason')"; + if json_has "choices.0.finish_reason"; then + if [ "x$(json_get "choices.0.finish_reason")" != "xstop" ]; then + logging_error "Unexpected finish reason: $(json_get "choices.0.finish_reason")"; else - echo "${response}" | jq -j -r '.choices[0].message.content'; + json_get "choices.0.message.content"; echo ""; - if (echo "${response}" | jq -e -r '.usage' > /dev/null); then - prompt_tokens=$(echo "${response}" | jq -j -r '.usage.prompt_tokens'); - completion_tokens=$(echo "${response}" | jq -j -r '.usage.completion_tokens'); - total_tokens=$(echo "${response}" | jq -j -r '.usage.total_tokens'); + if json_has "usage"; then + prompt_tokens=$(json_get "usage.prompt_tokens"); + completion_tokens=$(json_get "usage.completion_tokens"); + total_tokens=$(json_get "usage.total_tokens"); echo ''; logging_info "usage: prompt_tokens=${prompt_tokens}, completion_tokens=${completion_tokens}, total_tokens=${total_tokens}"; fi @@ -46,20 +50,24 @@ generate_completion() { elif echo "x${line}" | grep -e "^xdata: {" > /dev/null 2>&1; then # Data chunk received json_chunk=$(echo "${line}" | cut -c 6-); - if (echo "${json_chunk}" | jq -e -r '.choices[0].delta.content' > /dev/null); then - echo "${json_chunk}" | jq -j -r '.choices[0].delta.content'; + if ! json_parse "${json_chunk}"; then + logging_debug "Unexpected chunk: ${json_chunk}"; + continue; + fi + if json_has "choices.0.delta.content"; then + json_get "choices.0.delta.content"; else - if (echo "${json_chunk}" | jq -e -r '.finish_reason' > /dev/null); then - stop_reason=$(echo "${json_chunk}" | jq -j -r '.finish_reason'); + if json_has "finish_reason"; then + stop_reason=$(json_get "finish_reason"); if [ "x${stop_reason}" != "xstop" ]; then logging_error "Unexpected stop reason: ${stop_reason}"; fi break; - elif (echo "${json_chunk}" | jq -e -r '.usage' > /dev/null); then + elif json_has "usage"; then # Data chunk contains usage information (This is usually the last chunk) - prompt_tokens=$(echo "${json_chunk}" | jq -j -r '.usage.prompt_tokens'); - completion_tokens=$(echo "${json_chunk}" | jq -j -r '.usage.completion_tokens'); - total_tokens=$(echo "${json_chunk}" | jq -j -r '.usage.total_tokens'); + prompt_tokens=$(json_get "usage.prompt_tokens"); + completion_tokens=$(json_get "usage.completion_tokens"); + total_tokens=$(json_get "usage.total_tokens"); echo ''; logging_info "usage: prompt_tokens=${prompt_tokens}, completion_tokens=${completion_tokens}, total_tokens=${total_tokens}"; fi @@ -81,4 +89,4 @@ generate_completion() { fi } -export generate_completion; \ No newline at end of file +export generate_completion; diff --git a/tests/entry.sh b/tests/entry.sh index 823e2db..0875048 100644 --- a/tests/entry.sh +++ b/tests/entry.sh @@ -3,7 +3,7 @@ set -o posix; echo "Setting up prerequisites..."; -apk -q add jq curl perl; +apk -q add curl perl; echo "Installing ell...";