Files
el/elp/src/organ-read.el
T
bigmerge b7e2c580a8
El SDK CI - dev / build-and-test (pull_request) Successful in 6m28s
Add native speech synthesis and voice-imitation faculty
speech.el: formant/glottal integer DSP synthesis + voice-analyze-by-
imitation. voice-profile.el / voice-ingest.el: voice-profile plumbing.
accent.el: British-RP as an ingested transform-geometry (explicitly marked
provisional/citation-pending by its own comments). organ-read.el:
engram read-through for the speech organ. Includes demo/test drivers and
non-personal reference data (British-RP phonetics/lexicon derived data,
a public-domain LibriVox RP reference recording).

Deliberately excludes elp/data/live/ (raw recorded voice + face-photo
samples of the repo owner) and the will-*.{json,psv} derived voiceprint
files — personal biometric data that shouldn't be committed to a shared
repo without an explicit decision from the owner. Also excludes this
worktree's elp/src/surface-profile.el, which diverges from the copy in
other worktrees (agent-aaf04b0a9714c4070, main) — needs manual
reconciliation before landing, left out here to avoid silently picking a
version.
2026-08-15 14:27:52 -05:00

126 lines
4.4 KiB
EmacsLisp

// organ-read.el - Route the render's GEOMETRY READ through the ingest ORGAN's
// saved engram files (the coordinator's source of truth). For each file we
// engram_load() it, engram_scan_nodes_json(limit, offset) to get the node array,
// and cache each node's self-contained CONTENT string keyed by symbol. Because
// the cached value carries the numbers ("... f1=730 ..."), the cache SURVIVES the
// store being REPLACED by the next engram_load so we load+cache phonetics
// FIRST, then load+cache accent. The .psv path remains a fallback.
//
// engram_scan_nodes_json(limit, offset) takes NO query; it returns nodes
// salience-sorted, so limit must be >= node count and we filter client-side.
// (engram_search / engram_scan_nodes return len-5 garbage unused.)
// Find every occurrence of `marker` in the scan JSON; for each, cache
// sym -> a 150-char content window (enough to hold f1..amp). Duplicates from the
// node's "content" and "label" fields are harmless (first match wins on read).
fn organ_cache(j: String, marker: String, mlen: Int, win_len: Int, need: String) -> [String] {
let m: [String] = native_list_empty()
let jl: Int = str_len(j)
let off: Int = 0
while off < jl {
let rest: String = str_slice(j, off, jl)
let p: Int = str_index_of(rest, marker)
if p < 0 {
off = jl
} else {
let abs: Int = off + p
let win: String = str_slice(j, abs, abs + win_len)
let after: String = str_slice(win, mlen, str_len(win))
let sp: Int = str_index_of(after, " ")
let hasneed: Int = str_index_of(win, need)
if sp > 0 {
if hasneed >= 0 {
let sym: String = str_slice(after, 0, sp)
m = native_list_append(m, sym)
m = native_list_append(m, win)
}
}
off = abs + mlen
}
}
return m
}
// Load the phonetics organ file and cache sym -> content. mlen("phoneme ")=8.
fn organ_pmap(path: String) -> [String] {
let ok: Bool = engram_load(path)
if ok == false {
return native_list_empty()
}
let j: String = engram_scan_nodes_json(600, 0)
return organ_cache(j, "phoneme ", 8, 150, "f1=")
}
// Load the accent organ file and cache sym -> content. mlen("accent_target ")=14.
// Vowel overrides carry f1=..; the R rule carries drop_coda_r (need="=" matches
// both, i.e. any well-formed accent_target field).
fn organ_amap(path: String) -> [String] {
let ok: Bool = engram_load(path)
if ok == false {
return native_list_empty()
}
let j: String = engram_scan_nodes_json(600, 0)
return organ_cache(j, "accent_target ", 14, 90, "=")
}
// Vowel-set (categorical class) from the phonetics .psv class column.
fn organ_vset(path: String) -> [String] {
let content: String = fs_read(path)
let lines: [String] = str_split(content, "\n")
let nl: Int = native_list_len(lines)
let v: [String] = native_list_empty()
let li: Int = 0
while li < nl {
let line: String = native_list_get(lines, li)
let ok: Int = 1
if str_len(line) < 5 {
ok = 0
}
if ok == 1 {
if str_char_code(line, 0) == 35 {
ok = 0
}
}
if ok == 1 {
let f: [String] = str_split(line, "|")
if native_list_len(f) >= 12 {
if str_eq(native_list_get(f, 11), "vowel") {
v = native_list_append(v, native_list_get(f, 0))
}
}
}
li = li + 1
}
return v
}
// Word -> phoneme-sequence cache from lexicon.psv (engram-independent).
fn organ_lex(path: String) -> [String] {
let content: String = fs_read(path)
let lines: [String] = str_split(content, "\n")
let nl: Int = native_list_len(lines)
let m: [String] = native_list_empty()
let li: Int = 0
while li < nl {
let line: String = native_list_get(lines, li)
let ok: Int = 1
if str_len(line) < 3 {
ok = 0
}
if ok == 1 {
if str_char_code(line, 0) == 35 {
ok = 0
}
}
if ok == 1 {
let f: [String] = str_split(line, "|")
if native_list_len(f) >= 2 {
m = native_list_append(m, native_list_get(f, 0))
m = native_list_append(m, native_list_get(f, 1))
}
}
li = li + 1
}
return m
}