Files
el/elp/src/voice-ingest.el
T
bigmerge b7e2c580a8
El SDK CI - dev / build-and-test (pull_request) Successful in 6m28s
Add native speech synthesis and voice-imitation faculty
speech.el: formant/glottal integer DSP synthesis + voice-analyze-by-
imitation. voice-profile.el / voice-ingest.el: voice-profile plumbing.
accent.el: British-RP as an ingested transform-geometry (explicitly marked
provisional/citation-pending by its own comments). organ-read.el:
engram read-through for the speech organ. Includes demo/test drivers and
non-personal reference data (British-RP phonetics/lexicon derived data,
a public-domain LibriVox RP reference recording).

Deliberately excludes elp/data/live/ (raw recorded voice + face-photo
samples of the repo owner) and the will-*.{json,psv} derived voiceprint
files — personal biometric data that shouldn't be committed to a shared
repo without an explicit decision from the owner. Also excludes this
worktree's elp/src/surface-profile.el, which diverges from the copy in
other worktrees (agent-aaf04b0a9714c4070, main) — needs manual
reconciliation before landing, left out here to avoid silently picking a
version.
2026-08-15 14:27:52 -05:00

245 lines
10 KiB
EmacsLisp

// voice-ingest.el - The LIVE VOICE LOOP reshape + ingest-as-geometry.
//
// EL cannot read a binary WAV (fs_read NUL-truncates), so the thin-medium DSP
// extractor is periph's `voiceprint` (autocorr F0 + LPC formants), equivalent to
// our own voice_analyze. This module: (1) RESHAPE the voiceprint JSON (TEXT) into
// the organ voice-signature schema; (2) INGEST it as a GEOMETRY manifold in the
// engram and engram_save it to a file; (3) READ the target signature BACK from
// that geometry (engram_load + scan + filter), never from the json or a table.
// HONEST: this reaches for pitch + a coarse vocal-tract scale (kf). It is NOT a
// clone no glottal timbre, vowel-space, or articulation is captured.
fn parse_leading_int(s: String) -> Int {
let n: Int = str_len(s)
let i: Int = 0
let v: Int = 0
let started: Int = 0
while i < n {
let c: Int = str_char_code(s, i)
if c >= 48 {
if c <= 57 {
v = v * 10 + (c - 48)
started = 1
i = i + 1
} else {
i = n
}
} else {
if started == 1 {
i = n
} else {
i = i + 1
}
}
}
return v
}
// voiceprint JSON -> organ voice-signature source file; returns [f0,f0_end,kf,f1,f2,f3].
fn reshape_voiceprint(vppath: String, outjson: String) -> [Int] {
let j: String = fs_read(vppath)
let f0: Int = parse_uint_from(j, "f0_hz\":")
let fp: Int = str_index_of(j, "formants_hz")
let tail: String = str_slice(j, fp, fp + 120)
let br: Int = str_index_of(tail, "[")
let arr: String = str_slice(tail, br + 1, str_len(tail))
let f1: Int = parse_leading_int(arr)
let c1: Int = str_index_of(arr, ",")
let a2: String = str_slice(arr, c1 + 1, str_len(arr))
let f2: Int = parse_leading_int(a2)
let c2: Int = str_index_of(a2, ",")
let a3: String = str_slice(a2, c2 + 1, str_len(a2))
let f3: Int = parse_leading_int(a3)
let f0e: Int = f0 * 85 / 100
// derive kf honestly: coarse vocal-tract scale from the formant pattern
let t1: Int = 1000 * f1 / 500
let t2: Int = 1000 * f2 / 1500
let t3: Int = 1000 * f3 / 2500
let kf: Int = (t1 + t2 + t3) / 3
if kf < 800 {
kf = 800
}
if kf > 1400 {
kf = 1400
}
let js: String = "{\"dataset\":\"will-voice-signature\",\"primitive_type\":\"voice\",\"grounding\":\"measured\",\"provenance\":\"Will live 30s read 2026-08-15 (elp/data/live/will30_clean.wav, 27.0s) SUPERSEDES the coarse 10s sample; F0+formants via periph voiceprint (autocorr+LPC), averaged over his full vowel set. Still the 11-number average: no coarticulation/prosody. COARSE — pitch + vocal-tract scale, NOT a clone.\",\"records\":[{\"key\":\"will\",\"features\":{\"source\":\"live-mic\"},\"attributes\":{\"f0\":" + int_to_str(f0) + ",\"f0_end\":" + int_to_str(f0e) + ",\"kf\":" + int_to_str(kf) + ",\"f1\":" + int_to_str(f1) + ",\"f2\":" + int_to_str(f2) + ",\"f3\":" + int_to_str(f3) + "}}]}"
let okw: Bool = fs_write(outjson, js)
let r: [Int] = native_list_empty()
let r = native_list_append(r, f0)
let r = native_list_append(r, f0e)
let r = native_list_append(r, kf)
let r = native_list_append(r, f1)
let r = native_list_append(r, f2)
let r = native_list_append(r, f3)
return r
}
// Ingest the signature as a manifold (a set-hub + the will node + a member edge)
// and engram_save it to a reloadable file. grounding:measured self-declared.
fn ingest_voice(sig: [Int], savepath: String) -> Int {
let f0: Int = native_list_get(sig, 0)
let f0e: Int = native_list_get(sig, 1)
let kf: Int = native_list_get(sig, 2)
let f1: Int = native_list_get(sig, 3)
let f2: Int = native_list_get(sig, 4)
let f3: Int = native_list_get(sig, 5)
let hub: String = engram_node("voice-signature-set will grounding=measured src=periph-voiceprint", "VoiceSet", 90)
let cont: String = "voice will | f0=" + int_to_str(f0) + " f0_end=" + int_to_str(f0e) + " kf=" + int_to_str(kf) + " f1=" + int_to_str(f1) + " f2=" + int_to_str(f2) + " f3=" + int_to_str(f3) + " grounding=measured src=periph-voiceprint-30s supersedes=prior-voice-region prov=COARSE-pitch+tractscale-NOT-a-clone"
let id: String = engram_node(cont, "Voice", 90)
engram_connect(id, hub, 90, "member_of")
let oks: Bool = engram_save(savepath)
return 1
}
// READ the target voice back FROM the ingested geometry (engram_load + scan +
// client-filter for "voice will"). Returns [f0,f0_end,kf,f1,f2,f3] or empty.
fn load_voice(savepath: String) -> [Int] {
let ok: Bool = engram_load(savepath)
let r: [Int] = native_list_empty()
if ok == false {
return r
}
let j: String = engram_scan_nodes_json(200, 0)
let p: Int = str_index_of(j, "voice will ")
if p < 0 {
return r
}
let win: String = str_slice(j, p, p + 200)
let r = native_list_append(r, parse_uint_from(win, "f0="))
let r = native_list_append(r, parse_uint_from(win, "f0_end="))
let r = native_list_append(r, parse_uint_from(win, "kf="))
let r = native_list_append(r, parse_uint_from(win, "f1="))
let r = native_list_append(r, parse_uint_from(win, "f2="))
let r = native_list_append(r, parse_uint_from(win, "f3="))
return r
}
// ---- Vowel-space + prosody: ingest-as-geometry + read-back (no source layer) --
// vowel target lookup from the ingested vowel-space manifold: sym -> [f1,f2,f3].
fn vmap_get(vmap: [String], code: String) -> [Int] {
let out: [Int] = native_list_empty()
let id: String = sp_map_get(vmap, code)
if str_eq(id, "") {
return out
}
let f1: Int = parse_uint_from(id, "f1=")
if f1 <= 0 {
return out
}
let out = native_list_append(out, f1)
let out = native_list_append(out, parse_uint_from(id, "f2="))
let out = native_list_append(out, parse_uint_from(id, "f3="))
return out
}
// Ingest his measured vowel space + prosody as ONE manifold (VowelSpace hub +
// per-vowel target nodes + a prosody node) and engram_save it. Fresh empty store
// per run => set-replace, no duplicate.
fn ingest_voicegeom(vpath: String, ppath: String, savepath: String) -> Int {
let hub: String = engram_node("vowel-space-set will grounding=measured src=lpc-formant-track-30s", "VowelSpace", 90)
let content: String = fs_read(vpath)
let lines: [String] = str_split(content, "\n")
let nl: Int = native_list_len(lines)
let li: Int = 0
while li < nl {
let line: String = native_list_get(lines, li)
let ok: Int = 1
if str_len(line) < 5 {
ok = 0
}
if ok == 1 {
if str_char_code(line, 0) == 35 {
ok = 0
}
}
if ok == 1 {
let f: [String] = str_split(line, "|")
if native_list_len(f) >= 5 {
let sym: String = native_list_get(f, 0)
let cont: String = "vowel-target will " + sym + " | f1=" + native_list_get(f, 1) + " f2=" + native_list_get(f, 2) + " f3=" + native_list_get(f, 3) + " n=" + native_list_get(f, 4) + " grounding=measured src=lpc-formant-track-30s"
let id: String = engram_node(cont, "VowelTarget", 90)
engram_connect(id, hub, 90, "member_of")
}
}
li = li + 1
}
let pc: String = fs_read(ppath)
let plines: [String] = str_split(pc, "\n")
let pnl: Int = native_list_len(plines)
let pi: Int = 0
while pi < pnl {
let pl: String = native_list_get(plines, pi)
let ok2: Int = 1
if str_len(pl) < 5 {
ok2 = 0
}
if ok2 == 1 {
if str_char_code(pl, 0) == 35 {
ok2 = 0
}
}
if ok2 == 1 {
let pf: [String] = str_split(pl, "|")
if native_list_len(pf) >= 4 {
let pcont: String = "prosody will | f0_median=" + native_list_get(pf, 0) + " f0_min=" + native_list_get(pf, 1) + " f0_max=" + native_list_get(pf, 2) + " declination=" + native_list_get(pf, 3) + " src=f0-contour-30s"
let pid: String = engram_node(pcont, "Prosody", 90)
engram_connect(pid, hub, 90, "prosody_of")
}
}
pi = pi + 1
}
let oks: Bool = engram_save(savepath)
return 1
}
// Read the vowel-space back from geometry; prosody folded under key __PROSODY__.
fn load_voicegeom(savepath: String) -> [String] {
let m: [String] = native_list_empty()
let ok: Bool = engram_load(savepath)
if ok == false {
return m
}
let j: String = engram_scan_nodes_json(400, 0)
let jl: Int = str_len(j)
let off: Int = 0
while off < jl {
let rest: String = str_slice(j, off, jl)
let p: Int = str_index_of(rest, "vowel-target will ")
if p < 0 {
off = jl
} else {
let abs: Int = off + p
let win: String = str_slice(j, abs, abs + 140)
let after: String = str_slice(win, 18, str_len(win))
let sp: Int = str_index_of(after, " ")
if sp > 0 {
let sym: String = str_slice(after, 0, sp)
m = native_list_append(m, sym)
m = native_list_append(m, win)
}
off = abs + 18
}
}
let pp: Int = str_index_of(j, "prosody will ")
if pp >= 0 {
let pwin: String = str_slice(j, pp, pp + 160)
m = native_list_append(m, "__PROSODY__")
m = native_list_append(m, pwin)
}
return m
}
// Prosody stats [f0_median, f0_min, f0_max, declination] read from geometry.
fn prosody_from(vmap: [String]) -> [Int] {
let out: [Int] = native_list_empty()
let id: String = sp_map_get(vmap, "__PROSODY__")
if str_eq(id, "") {
return out
}
let out = native_list_append(out, parse_uint_from(id, "f0_median="))
let out = native_list_append(out, parse_uint_from(id, "f0_min="))
let out = native_list_append(out, parse_uint_from(id, "f0_max="))
let out = native_list_append(out, parse_uint_from(id, "declination="))
return out
}