Devices & Components
1
Arduino® UNO™ Q 2GB
Software & Tools
Arduino App Lab
Project description
Code
main
python
The main Python file to interface with Llama
1""" 2Termi3 - Python Script 3Cameron Coward 4cameroncoward.com 59/24/2026 6""" 7import http.client 8import json 9import re 10import socket 11import threading 12 13from arduino.app_utils import * # noqa: F401,F403 (Bridge, App) 14 15# The llama-server runs as a systemd service on the HOST (bound to 0.0.0.0:8080). 16# This Python app runs inside an App Lab Docker container, so "127.0.0.1" is the 17# container's own loopback — NOT the host. We must reach the host via the Docker 18# gateway. Try the standard host aliases / gateway IPs in order and pick the 19# first that resolves+connects. 20LLM_PORT = 8080 21_HOST_CANDIDATES = [ 22 "host.docker.internal", # Docker's standard host alias (if enabled) 23 "172.19.0.1", # termi3_default network gateway (observed) 24 "172.17.0.1", # default docker0 bridge gateway (fallback) 25 "127.0.0.1", # last resort (works only if not containerized) 26] 27 28 29def _pick_llm_host() -> str: 30 for host in _HOST_CANDIDATES: 31 try: 32 with socket.create_connection((host, LLM_PORT), timeout=2): 33 print(f"[termi3] llama-server reachable at {host}:{LLM_PORT}") 34 return host 35 except OSError: 36 continue 37 # Nothing reachable yet; default to the observed gateway and let requests 38 # surface the error (server may still be starting). 39 print("[termi3] WARNING: no llama-server host reachable yet; using gateway") 40 return "172.19.0.1" 41 42 43LLM_HOST = _pick_llm_host() 44LLM_PATH = "/v1/chat/completions" 45 46# Reuse a single HTTP connection across questions (avoids a fresh TCP handshake 47# per request). Guarded by a lock since Bridge calls may arrive on a thread. 48_conn_lock = threading.Lock() 49_conn = None 50 51 52def _get_conn(): 53 global _conn 54 if _conn is None: 55 _conn = http.client.HTTPConnection(LLM_HOST, LLM_PORT, timeout=HTTP_TIMEOUT) 56 return _conn 57 58 59def _reset_conn(): 60 global _conn 61 try: 62 if _conn is not None: 63 _conn.close() 64 except Exception: # noqa: BLE001 65 pass 66 _conn = None 67 68# Concise/factual system prompt (no personality). Kept SHORT on purpose: fewer 69# prompt tokens = faster prompt processing and a smaller, more stable prefix 70# cache. Measured ~2.1-2.7s time-to-first-token vs ~5-10s with the long prompt. 71SYSTEM_PROMPT = ( 72 "Answer the question concisely and factually in as few words as possible, " 73 "ideally one short sentence. No commentary or extra words." 74) 75 76MAX_TOKENS = 64 77TEMPERATURE = 0.3 78HTTP_TIMEOUT = 30 # seconds; the MCU-side Bridge timeout should exceed this 79 80# --- Output sanitization for the Silent 700 (uppercase ASCII printer) ------- 81_MD = re.compile(r"[*_`#>]+") 82_UNICODE = { 83 "\u2019": "'", "\u2018": "'", "\u201c": '"', "\u201d": '"', 84 "\u2013": "-", "\u2014": "-", "\u2026": "...", "\u00a0": " ", 85} 86 87 88def sanitize(text: str) -> str: 89 text = _MD.sub("", text) 90 for k, v in _UNICODE.items(): 91 text = text.replace(k, v) 92 text = text.encode("ascii", "ignore").decode("ascii") 93 # Collapse whitespace/newlines into single spaces (printer handles CR/LF 94 # via the MCU; the answer body should be one clean line). 95 text = " ".join(text.split()) 96 return text.strip() 97 98 99def _query_llm(question: str) -> str: 100 body = { 101 "messages": [ 102 {"role": "system", "content": SYSTEM_PROMPT}, 103 {"role": "user", "content": question}, 104 ], 105 "max_tokens": MAX_TOKENS, 106 "temperature": TEMPERATURE, 107 "cache_prompt": True, 108 } 109 data = json.dumps(body).encode() 110 headers = {"Content-Type": "application/json"} 111 112 with _conn_lock: 113 # One retry: if the persisted connection went stale, reset and redo. 114 for attempt in (1, 2): 115 try: 116 conn = _get_conn() 117 conn.request("POST", LLM_PATH, body=data, headers=headers) 118 resp = conn.getresponse() 119 raw = resp.read() 120 payload = json.loads(raw) 121 return payload["choices"][0]["message"]["content"] 122 except (http.client.HTTPException, OSError): 123 _reset_conn() 124 if attempt == 2: 125 raise 126 127 128def ask(question: str) -> str: 129 """Called by the MCU over the Bridge. Returns a printable answer line.""" 130 q = (question or "").strip() 131 if not q: 132 return "NO QUESTION RECEIVED." 133 try: 134 raw = _query_llm(q) 135 answer = sanitize(raw) 136 if not answer: 137 answer = "NO ANSWER." 138 print(f"[termi3] Q: {q!r} -> A: {answer!r}") # App Lab console log 139 return answer 140 except Exception as exc: # noqa: BLE001 141 print(f"[termi3] error: {exc}") 142 return "SORRY, MY BRAIN IS OFFLINE." 143 144 145def _llm_health_ok() -> bool: 146 """True if llama-server answers /health with 200.""" 147 try: 148 with _conn_lock: 149 conn = _get_conn() 150 conn.request("GET", "/health") 151 resp = conn.getresponse() 152 resp.read() 153 return resp.status == 200 154 except (http.client.HTTPException, OSError): 155 _reset_conn() 156 return False 157 158 159def _warm_cache() -> None: 160 """Prime the llama-server prefix cache with our exact system prompt so the 161 first REAL question gets the fast (~2s) path instead of a cold (~10s) one. 162 """ 163 try: 164 _query_llm("ping") 165 print("[termi3] prefix cache warmed") 166 except Exception as exc: # noqa: BLE001 167 print(f"[termi3] warm-up skipped: {exc}") 168 169 170_warmed = False 171 172 173def is_ready() -> bool: 174 """MCU polls this at startup to know Python is up. 175 176 On a cold boot the llama-server (separate systemd service) may still be 177 loading its model. To make the FIRST question reliable, we don't report 178 ready until the LLM answers /health, then we warm the cache once. 179 The MCU keeps polling, so returning False just makes it wait. 180 """ 181 global _warmed 182 if _warmed: 183 return True 184 if not _llm_health_ok(): 185 return False # LLM not up yet; MCU will poll again 186 _warmed = True 187 _warm_cache() 188 return True 189 190 191# Register functions the MCU can call. 192Bridge.provide("ask", ask) # noqa: F405 193Bridge.provide("is_ready", is_ready) # noqa: F405 194 195 196def loop(): 197 # Nothing to do in the main loop; all work is driven by MCU calls. 198 import time 199 time.sleep(1) 200 201 202App.run(user_loop=loop) # noqa: F405
sketch
cpp
The main Arduino sketch
1/* 2 * """ 3 * Termi3 - Python Script 4 * Cameron Coward 5 * cameroncoward.com 6 * 9/24/2026 7 * """ 8 * 9 * 1. Print "ASK TERMI3: " (bit-bang TX) 10 * 2. Read keystrokes until CR (bit-bang RX -> line buffer) 11 * 3. Bridge.call("ask", question) (blocks for the answer) 12 * 4. Print "TERMI3: " + answer (bit-bang TX) 13 * 5. Loop 14 * 15 * serial config (Phase 3): A0=TX, A1=RX, 300 baud (BIT_US=3333), 16 * 7 data bits, odd parity, 1 stop, TX_INVERT=false / RX idle HIGH. 17 * Terminal switches: HIGH SPEED, HALF DUPLEX, ON LINE. 18 */ 19 20#include <Arduino_RouterBridge.h> 21#include <Arduino_LED_Matrix.h> 22 23// ---------------- Status LED matrix (UNO Q, 8x13) ---------------- 24// Three status icons show where the loop is, set once at each boundary 25// (never refreshed in loop()). The firmware refresh IRQ keeps the frame lit. 26// NOTE: that refresh IRQ runs during bit-bang serial. If it ever jitters the 27// 3.3ms/bit timing (corrupt chars), matrix.end() disables it as a fallback. 28Arduino_LED_Matrix matrix; 29 30// 8 rows x 13 cols, row-major, 1 = LED on. setGrayscaleBits(1) -> on/off. 31 32// "?" — waiting for a question 33const uint8_t ICON_WAIT[104] = { 34 0,0,0,0,1,1,1,1,1,0,0,0,0, 35 0,0,0,1,1,0,0,0,1,1,0,0,0, 36 0,0,0,0,0,0,0,0,1,1,0,0,0, 37 0,0,0,0,0,0,0,1,1,0,0,0,0, 38 0,0,0,0,0,0,1,1,0,0,0,0,0, 39 0,0,0,0,0,0,1,1,0,0,0,0,0, 40 0,0,0,0,0,0,0,0,0,0,0,0,0, 41 0,0,0,0,0,0,1,1,0,0,0,0,0 42}; 43 44// hourglass — processing (waiting on the LLM answer) 45const uint8_t ICON_PROC[104] = { 46 0,0,0,1,1,1,1,1,1,1,0,0,0, 47 0,0,0,0,1,1,1,1,1,0,0,0,0, 48 0,0,0,0,0,1,1,1,0,0,0,0,0, 49 0,0,0,0,0,0,1,0,0,0,0,0,0, 50 0,0,0,0,0,0,1,0,0,0,0,0,0, 51 0,0,0,0,0,1,1,1,0,0,0,0,0, 52 0,0,0,0,1,1,1,1,1,0,0,0,0, 53 0,0,0,1,1,1,1,1,1,1,0,0,0 54}; 55 56// speaker — responding (printing the answer) 57const uint8_t ICON_RESP[104] = { 58 0,0,0,0,0,1,0,0,0,0,0,0,0, 59 0,0,0,0,1,1,0,0,1,0,0,0,0, 60 0,0,1,1,1,1,0,1,0,1,0,0,0, 61 0,0,1,1,1,1,0,1,0,1,0,1,0, 62 0,0,1,1,1,1,0,1,0,1,0,1,0, 63 0,0,1,1,1,1,0,1,0,1,0,0,0, 64 0,0,0,0,1,1,0,0,1,0,0,0,0, 65 0,0,0,0,0,1,0,0,0,0,0,0,0 66}; 67 68// A0/A1 (STM32U585 PA4/PA5, aliased D14/D15) are full digital GPIO — the 69// bit-bang is pure polled digitalRead/Write, so no pin-specific peripheral is 70// needed. NOTE: A0/A1 are NOT 5V tolerant; the MAX3232 runs at 3.3V so R1OUT 71// swings 0-3.3V. Keep the 5V connect-detect line (DA-15 pin 11) away from them. 72const int PIN_TX = A0; 73const int PIN_RX = A1; 74const int BIT_US = 3333; 75const int HALF_BIT_US = 1666; 76const bool TX_INVERT = false; 77const bool RX_IDLE_HIGH = true; 78 79const int MAX_Q = 200; // max question length 80 81// ---------------- TX (confirmed) ---------------- 82inline void txBit(bool level) { 83 digitalWrite(PIN_TX, TX_INVERT ? !level : level); 84} 85void sendChar(char c) { 86 uint8_t data = (uint8_t)c & 0x7F; 87 txBit(false); delayMicroseconds(BIT_US); // start 88 uint8_t ones = 0; 89 for (int i = 0; i < 7; i++) { 90 bool b = (data >> i) & 0x01; if (b) ones++; 91 txBit(b); delayMicroseconds(BIT_US); 92 } 93 txBit(ones % 2 == 0); delayMicroseconds(BIT_US); // odd parity 94 txBit(true); delayMicroseconds(BIT_US); // stop 95} 96void sendString(const char *s) { while (*s) sendChar(*s++); } 97void sendString(const String &s) { 98 for (unsigned i = 0; i < s.length(); i++) sendChar(s[i]); 99} 100void newline() { sendChar('\r'); sendChar('\n'); } 101// Advance the paper n lines (CR+LF each — bare CR would not feed the paper). 102void feedLines(int n) { while (n-- > 0) newline(); } 103 104// ---------------- RX (confirmed) ---------------- 105inline bool rxLevel() { 106 bool high = (digitalRead(PIN_RX) == HIGH); 107 return RX_IDLE_HIGH ? high : !high; 108} 109// Blocking read of one char. Returns -1 on framing glitch. 110int receiveChar() { 111 while (rxLevel() == true) { /* idle */ } 112 delayMicroseconds(HALF_BIT_US); 113 if (rxLevel() != false) return -1; // false start 114 delayMicroseconds(BIT_US); 115 uint8_t data = 0; 116 for (int i = 0; i < 7; i++) { 117 if (rxLevel()) data |= (1 << i); 118 delayMicroseconds(BIT_US); 119 } 120 delayMicroseconds(BIT_US); // past parity 121 return data & 0x7F; 122} 123 124// Read a full line until CR (0x0D) or LF, into buf. Returns length. 125// Uppercases letters (terminal is uppercase anyway; keeps things clean). 126int readLine(char *buf, int maxlen) { 127 int n = 0; 128 for (;;) { 129 int c = receiveChar(); 130 if (c < 0) continue; // ignore framing glitches 131 if (c == '\r' || c == '\n') break; // end of question 132 if (c == 8 || c == 127) { // backspace/delete 133 if (n > 0) n--; 134 continue; 135 } 136 if (n < maxlen - 1 && c >= 32) { // printable 137 buf[n++] = (char)c; 138 } 139 } 140 buf[n] = '\0'; 141 return n; 142} 143 144// ---------------- App flow ---------------- 145char question[MAX_Q]; 146 147void printPrompt() { 148 newline(); 149 sendString("ASK TERMI3: "); 150} 151 152void setup() { 153 pinMode(PIN_TX, OUTPUT); 154 txBit(true); // idle mark 155 pinMode(PIN_RX, INPUT); 156 157 matrix.begin(); // starts firmware refresh IRQ 158 matrix.setGrayscaleBits(1); // 1 bit -> on/off 159 matrix.draw(ICON_PROC); // "processing" while warming up 160 161 Bridge.begin(); // ~2s, blocking 162 Monitor.begin(115200); // debug to App Lab console 163 164 // Wait until the Python side AND the local LLM are ready. is_ready() only 165 // returns true once llama-server answers /health, so on a cold boot this 166 // loop waits out the model-load time. Wait indefinitely (conference: the 167 // board is powered on and must self-heal, never give up). 168 bool ready = false; 169 unsigned long lastDot = 0; 170 while (!ready) { 171 Bridge.call("is_ready").result(ready); 172 if (!ready) { 173 // Print a slow row of dots so the paper shows it's warming up, not dead. 174 unsigned long now = millis(); 175 if (now - lastDot > 3000) { sendChar('.'); lastDot = now; } 176 delay(200); 177 } 178 } 179 180 delay(300); 181 newline(); 182 // Startup banner (branded prompt, per on-paper interaction design). 183 sendString("TERMI3 - A TYPEWRITER THAT THINKS"); 184 newline(); 185 186 matrix.draw(ICON_WAIT); // ready, waiting for a question 187} 188 189void loop() { 190 matrix.draw(ICON_WAIT); // waiting for a question 191 printPrompt(); 192 193 int len = readLine(question, MAX_Q); 194 if (len == 0) return; // empty line -> re-prompt 195 196 matrix.draw(ICON_PROC); // CR received -> processing 197 198 // Roll the paper up so the user can read back what they just typed. 199 feedLines(4); 200 201 // Static "thinking" marker. Printed BEFORE the blocking Bridge call, so it 202 // appears the moment the question is accepted. This is an ACKNOWLEDGEMENT, 203 // not a progress bar: at the conference many users couldn't tell whether 204 // their question had been received. A timed/animated indicator isn't 205 // possible here anyway — Bridge.call().result() blocks, so loop() gets no 206 // chance to print while waiting (see project memory: no concurrency, to 207 // protect bit-bang timing). 208 sendString("..."); 209 210 // Ask Python (which queries the local LLM). Blocks for the answer. 211 // NOTE: using the plain .result() form (confirmed by docs). Timeout is 212 // handled Python-side (urllib 30s). If the default Bridge timeout proves 213 // too short for a cold first answer, we'll add MCU-side timeout config. 214 String answer; 215 bool ok = Bridge.call("ask", String(question)).result(answer); 216 217 feedLines(2); // separate the marker from the answer 218 219 matrix.draw(ICON_RESP); // answer in hand -> responding 220 sendString("TERMI3: "); 221 if (ok && answer.length() > 0) { 222 sendString(answer); 223 } else { 224 sendString("SORRY, NO ANSWER."); 225 } 226 227 feedLines(4); // leave the answer visible above the platen 228}
Downloadable files
termi3_pcb
PCB files
termi3_pcb.zip
Comments
Only logged in users can leave comments