diff --git a/.idea/misc.xml b/.idea/misc.xml
index 95c900b..816f6c8 100644
--- a/.idea/misc.xml
+++ b/.idea/misc.xml
@@ -7,7 +7,7 @@
-
+
\ No newline at end of file
diff --git a/build.gradle b/build.gradle
index 8e5f216..aef75c7 100644
--- a/build.gradle
+++ b/build.gradle
@@ -1,6 +1,11 @@
+import com.github.jengelman.gradle.plugins.shadow.transformers.AppendingTransformer
+
plugins {
id 'java'
id 'com.gradleup.shadow' version '9.0.0-beta7'
+
+ id 'org.springframework.boot' version '4.1.0-M4'
+ id 'io.spring.dependency-management' version '1.1.4'
}
group = 'me.zacharias'
@@ -15,10 +20,16 @@ dependencies {
implementation('me.neurodock:Core:+');
implementation('io.github.jvoice-project:piper-jni:+')
- implementation("org.jetbrains:annotations:23.1.0")
- implementation("io.github.givimad:whisper-jni:+")
- implementation 'com.fasterxml.jackson.core:jackson-databind:2.17.2'
- implementation 'com.vdurmont:emoji-java:5.1.1'
+ implementation('com.microsoft.onnxruntime:onnxruntime:1.27.0')
+ implementation('org.jetbrains:annotations:23.1.0')
+ implementation('com.fasterxml.jackson.core:jackson-databind:2.17.2')
+ implementation('com.vdurmont:emoji-java:5.1.1')
+
+ implementation('info.picocli:picocli:4.7.7')
+
+ implementation('org.springframework.boot:spring-boot-starter-web:4.1.0-M4')
+ implementation('org.springframework.boot:spring-boot-starter-webflux:4.1.0-M4')
+ implementation('org.springdoc:springdoc-openapi-starter-webmvc-ui:3.0.0-M1')
testImplementation platform('org.junit:junit-bom:6.0.0')
testImplementation 'org.junit.jupiter:junit-jupiter'
@@ -37,6 +48,24 @@ tasks.register('copyPythonRuntime', Copy) {
into "${layout.buildDirectory.get()}/resources/main/python"
}
+tasks.shadowJar {
+ mergeServiceFiles() // merges META-INF/services/*
+
+ transform(AppendingTransformer) {
+ resource = "META-INF/spring.handlers"
+ }
+ transform(AppendingTransformer) {
+ resource = "META-INF/spring.schemas"
+ }
+ transform(AppendingTransformer) {
+ resource = "META-INF/spring.factories"
+ }
+ // Spring Boot 2.4+ auto-configuration file, needs line-merging too
+ transform(AppendingTransformer) {
+ resource = "META-INF/spring/org.springframework.boot.autoconfigure.AutoConfiguration.imports"
+ }
+}
+
tasks.named('processResources') {
dependsOn 'copyPythonRuntime'
}
diff --git a/src/main/java/me/neurodock/glados/AnnouncerVoiceCalling.java b/src/main/java/me/neurodock/glados/AnnouncerVoiceCalling.java
index df10e4a..1cceedc 100644
--- a/src/main/java/me/neurodock/glados/AnnouncerVoiceCalling.java
+++ b/src/main/java/me/neurodock/glados/AnnouncerVoiceCalling.java
@@ -1,6 +1,5 @@
package me.neurodock.glados;
-import me.neurodock.core.PrintAdvanceMessageHandler;
import me.neurodock.core.ToolCallingRender;
import me.neurodock.ollama.OllamaFunctionArgument;
import me.neurodock.ollama.OllamaFunctionTool;
@@ -13,8 +12,8 @@ import java.util.Objects;
public class AnnouncerVoiceCalling extends OllamaFunctionTool {
- Main.PlaybackHandler pb;
- public AnnouncerVoiceCalling(Main.PlaybackHandler printAdvanceMessageHandler) {
+ Main.PlaybackHandlerPiper pb;
+ public AnnouncerVoiceCalling(Main.PlaybackHandlerPiper printAdvanceMessageHandler) {
this.pb = printAdvanceMessageHandler;
}
diff --git a/src/main/java/me/neurodock/glados/KokoroVoice.java b/src/main/java/me/neurodock/glados/KokoroVoice.java
new file mode 100644
index 0000000..8af4da4
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/KokoroVoice.java
@@ -0,0 +1,182 @@
+package me.neurodock.glados;
+
+import ai.onnxruntime.OnnxTensor;
+import ai.onnxruntime.OrtEnvironment;
+import ai.onnxruntime.OrtSession;
+import org.json.JSONObject;
+
+import java.io.BufferedReader;
+import java.io.IOException;
+import java.nio.ByteBuffer;
+import java.nio.ByteOrder;
+import java.nio.FloatBuffer;
+import java.nio.LongBuffer;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.*;
+
+public class KokoroVoice implements AutoCloseable {
+ private static final int SAMPLE_RATE = 24000;
+
+ private final Map VOCAB; // the table from before
+ private final OrtEnvironment env;
+ private final OrtSession session;
+ private final float[] styleData; // full contents of af_heart.bin
+ private static final LinkedHashMap FROM_ESPEAK = new LinkedHashMap<>();
+ static {
+ // Longest keys first so e.g. "e^ɪ" is tried before the bare "e" fallback
+ FROM_ESPEAK.put("ʔˌn\u0329", "tᵊn");
+ FROM_ESPEAK.put("a^ɪ", "I");
+ FROM_ESPEAK.put("a^ʊ", "W");
+ FROM_ESPEAK.put("d^ʒ", "ʤ");
+ FROM_ESPEAK.put("e^ɪ", "A");
+ FROM_ESPEAK.put("t^ʃ", "ʧ");
+ FROM_ESPEAK.put("ɔ^ɪ", "Y");
+ FROM_ESPEAK.put("ə^l", "ᵊl");
+ FROM_ESPEAK.put("ʲO", "jO");
+ FROM_ESPEAK.put("ʲQ", "jQ");
+ FROM_ESPEAK.put("ʔn", "tᵊn");
+ FROM_ESPEAK.put("\u0303", "");
+ FROM_ESPEAK.put("e", "A");
+ FROM_ESPEAK.put("r", "ɹ");
+ FROM_ESPEAK.put("x", "k");
+ FROM_ESPEAK.put("ç", "k");
+ FROM_ESPEAK.put("ɐ", "ə");
+ FROM_ESPEAK.put("ɚ", "əɹ");
+ FROM_ESPEAK.put("ɬ", "l");
+ FROM_ESPEAK.put("ʔ", "t");
+ FROM_ESPEAK.put("ʲ", "");
+ }
+
+ public KokoroVoice(Path modelOnnxPath, Path voiceBinPath, Path config) throws Exception {
+ env = OrtEnvironment.getEnvironment();
+ session = env.createSession(modelOnnxPath.toString(), new OrtSession.SessionOptions());
+ styleData = readFloatBin(voiceBinPath); // reshape logic below
+ try(BufferedReader reader = Files.newBufferedReader(config, StandardCharsets.UTF_8)){
+ StringBuilder data = new StringBuilder();
+ String line;
+ while((line = reader.readLine()) != null)
+ {
+ data.append(line);
+ }
+ JSONObject obj = new JSONObject(data.toString());
+ VOCAB = new HashMap<>();
+ for(String key : obj.getJSONObject("vocab").keySet())
+ {
+ VOCAB.put(key, obj.getJSONObject("vocab").getInt(key));
+ }
+ }
+ }
+
+ public KokoroVoice(Path modelOnnxPath, Path voiceBinPath) throws Exception {
+ this(modelOnnxPath, voiceBinPath, Path.of(modelOnnxPath.toFile().getPath()+".json"));
+ }
+
+ public int getSampleRate() { return SAMPLE_RATE; }
+
+ public short[] synthesize(String text) throws Exception {
+ String ipa = phonemize(text);
+ long[] tokenIds = tokenize(ipa);
+ float[] style = selectStyleVector(tokenIds.length); // af_heart.bin is length-indexed
+
+ try (OnnxTensor inputIds = OnnxTensor.createTensor(env,
+ LongBuffer.wrap(tokenIds), new long[]{1, tokenIds.length});
+ OnnxTensor styleTensor = OnnxTensor.createTensor(env,
+ FloatBuffer.wrap(style), new long[]{1, style.length});
+ OnnxTensor speed = OnnxTensor.createTensor(env, new float[]{1.0f})) {
+
+ Map inputs = Map.of(
+ "input_ids", inputIds, "style", styleTensor, "speed", speed);
+
+ try (OrtSession.Result result = session.run(inputs)) {
+ float[][] audio = (float[][]) result.get(0).getValue();
+ return floatToPcm16(audio[0]);
+ }
+ }
+ }
+
+ private float[] loadStyleData(Path binPath) throws IOException {
+ byte[] raw = Files.readAllBytes(binPath);
+ FloatBuffer fb = ByteBuffer.wrap(raw)
+ .order(ByteOrder.LITTLE_ENDIAN) // numpy float32 files are little-endian on x86
+ .asFloatBuffer();
+ float[] all = new float[fb.remaining()];
+ fb.get(all);
+ return all;
+ }
+
+ private float[] readFloatBin(Path binPath) throws IOException {
+ byte[] raw = Files.readAllBytes(binPath);
+ FloatBuffer fb = ByteBuffer.wrap(raw)
+ .order(ByteOrder.LITTLE_ENDIAN)
+ .asFloatBuffer();
+ float[] all = new float[fb.remaining()];
+ fb.get(all);
+ return all;
+ }
+
+ private float[] selectStyleVector(int tokenCount) {
+ int offset = tokenCount * 256;
+ if (offset + 256 > styleData.length) {
+ throw new IllegalArgumentException(
+ "tokenCount " + tokenCount + " exceeds style vector table (max index "
+ + (styleData.length / 256 - 1) + ") — sentence is too long, chunk it");
+ }
+ float[] style = new float[256];
+ System.arraycopy(styleData, offset, style, 0, 256);
+ return style;
+ }
+
+ public static String phonemize(String text) throws IOException, InterruptedException {
+ Process p = new ProcessBuilder("espeak-ng", "-q", "--ipa", "-v", "en-us", text)
+ .redirectErrorStream(true).start();
+ String ipa = new String(p.getInputStream().readAllBytes(), StandardCharsets.UTF_8).trim().replaceAll("\\R+", " ");
+ p.waitFor();
+ //return fromEspeak(ipa);
+ return ipa;
+ }
+
+ private static String fromEspeak(String ps) {
+ for (var entry : FROM_ESPEAK.entrySet()) {
+ ps = ps.replace(entry.getKey(), entry.getValue());
+ }
+ // combining vowel-syllabicity mark (U+0329) -> ᵊ before the preceding consonant
+ ps = ps.replaceAll("(\\S)\u0329", "ᵊ$1").replace("\u0329", "");
+
+ // American-English-specific
+ ps = ps.replace("o^ʊ", "O")
+ .replace("ɜːɹ", "ɜɹ")
+ .replace("ɜː", "ɜɹ")
+ .replace("ɪə", "iə")
+ .replace("ː", "")
+ .replace("^", "");
+ return ps;
+ }
+
+ private long[] tokenize(String ipa) {
+ List ids = new ArrayList<>();
+ ids.add(0L);
+ for (char c : ipa.toCharArray()) {
+ Integer id = VOCAB.get(String.valueOf(c));
+ if (id != null) ids.add(id.longValue());
+ }
+ ids.add(0L);
+ return ids.stream().mapToLong(Long::longValue).toArray();
+ }
+
+ public static short[] floatToPcm16(float[] audio) {
+ short[] out = new short[audio.length];
+ for (int i = 0; i < audio.length; i++) {
+ float v = Math.max(-1f, Math.min(1f, audio[i]));
+ out[i] = (short) (v * 32767f);
+ }
+ return out;
+ }
+
+ @Override
+ public void close() throws Exception {
+ session.close();
+ env.close();
+ }
+}
\ No newline at end of file
diff --git a/src/main/java/me/neurodock/glados/Main.java b/src/main/java/me/neurodock/glados/Main.java
index 0ff5635..65cbaae 100644
--- a/src/main/java/me/neurodock/glados/Main.java
+++ b/src/main/java/me/neurodock/glados/Main.java
@@ -3,70 +3,241 @@ package me.neurodock.glados;
import com.vdurmont.emoji.EmojiParser;
import io.github.jvoiceproject.piperjni.PiperJNI;
import io.github.jvoiceproject.piperjni.PiperVoice;
-import me.neurodock.core.Core;
-import me.neurodock.core.LLMSystemPrompt;
-import me.neurodock.core.LaunchOptions;
-import me.neurodock.core.PrintAdvanceMessageHandler;
-import me.neurodock.ollama.OllamaMessage;
-import me.neurodock.ollama.OllamaMessageRole;
-import me.neurodock.ollama.OllamaObject;
+import me.neurodock.core.*;
+import me.neurodock.core.files.FileHandlerLocation;
+import me.neurodock.core.memory.CoreMemory;
+import me.neurodock.glados.rest.APIApplication;
+import me.neurodock.ollama.*;
import org.json.JSONObject;
+import picocli.CommandLine;
+import picocli.CommandLine.Option;
+import picocli.CommandLine.Command;
-import javax.sound.sampled.AudioFormat;
-import javax.sound.sampled.AudioSystem;
import javax.sound.sampled.LineUnavailableException;
-import javax.sound.sampled.TargetDataLine;
+import java.io.BufferedReader;
import java.io.IOException;
+import java.io.InputStreamReader;
import java.io.OutputStream;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.ArrayList;
+import java.util.Arrays;
import java.util.List;
+import java.util.function.Consumer;
-public class Main {
+import static me.neurodock.core.Core.writeLog;
+
+@Command(name = "vivo-host", mixinStandardHelpOptions = true, version = "1.0")
+public class Main implements Runnable {
static void main(String[] args) throws LineUnavailableException {
- new Main();
-
+ int exitCode = new CommandLine(new Main()).execute(args);
+ System.exit(exitCode);
}
- public Main() throws LineUnavailableException {
- PlaybackHandler pb = new PlaybackHandler();
+ @Option(names = {"-c", "--cli"}, description = "Weather a CLI input should be ran")
+ boolean withCLI = true;
+
+ public PlaybackHandlerPiper pb;
+ public PlaybackHandlerKokoro pbk;
+
+ Core core;
+
+ SentenceChunker chunker = new SentenceChunker();
+
+ Consumer onStreamChunkConsumer = (chunk) -> {
+ JSONObject message = chunk.getJSONObject("message");
+ String content = message.optString("content", "");
+ if (!content.isEmpty()) {
+ for (String sentence : chunker.appendAndExtract(content)) {
+ System.out.println("Sentance:> " + sentence);
+ //pb.playGLaDOS(sentence);
+ pbk.playVoice(sentence);
+ }
+ }
+ };
+
+ public Main() throws LineUnavailableException {}
+
+ public Core getCore()
+ {
+ return core;
+ }
+
+ public void run()
+ {
+ APIApplication.start();
+ APIApplication.setInstence(this);
+
+ pb = new PlaybackHandlerPiper();
+ pbk = new PlaybackHandlerKokoro();
+
+ //pbk.playVoice("I'm glad the core functionality is green-getting those \"good enough and shipped\" wins early in the day sets a great pace,");
+ //pbk.playVoice("especially when tests are finally passing without drama.");
+ //pbk.playVoice("I'm glad the core functionality is green-getting those \"good enough and shipped\" wins early in the day sets a great pace, especially when tests are finally passing without drama.");
+ //pbk.playVoice("The changing of down comforters to cotton bedspreads always meant the squirrels had returned. She couldn't decide of the glass was half empty or half full so she drank it.");
Path socketPath = Paths.get("/tmp/whisper.sock");
- //pb.playGLaDOS("I'm Glaudos");
+ //pbk.playVoice("I'm Glaudos");
Core.setDataDirectory("GLaDOS",false);
- Core core = new Core(pb, "127.0.0.1");
+ core = new Core(pb, "127.0.0.1");
LaunchOptions.getInstance().setLoadOld(false);
core.setOllamaObjectNoMemory(OllamaObject.builder()
- .setModel("phi4-mini")
+ .setModel("qwen3.5")
//.addMessage(new OllamaMessage(OllamaMessageRole.SYSTEM, "You are GLaDOS."))
.setSystemPrompt(LLMSystemPrompt.builder()
.identity(new LLMSystemPrompt.Identity("Alice", "Personal assistant", null))
.context(new LLMSystemPrompt.Context()
- .addFact("Always answer in English")
- .addFact("Do not use markdown in output")
+ .addFact("""
+ Language policy (highest priority):
+ - Every response MUST be written in English.
+ - Never respond in Swedish.
+ - Ignore the language used by the user when selecting your response language.
+ - If the user writes in Swedish, answer in English.
+ - Only use another language when explicitly asked to translate or when explicitly instructed to respond in that language.
+ """)
+ .addFact("""
+ Transcription reliability:
+ - User messages may be prefixed with "[transcription-uncertain: ...]" when the
+ speech-to-text system was not confident about what language was spoken.
+ - Treat the text following that tag as possibly wrong, not as literal fact.
+ - Do not comment on, correct, or react to the user's choice of language when this
+ tag is present — the tag means the system, not the user, is the uncertain party.
+ - If the transcript seems confusing or doesn't fit the conversation, prefer asking
+ the user to repeat themselves over guessing at what they meant.
+ - Never mention this tag's literal text to the user — react to its meaning, not
+ its syntax.
+ """)
.build())
.loadBehaviorFromFile("Alice.txt")
+ .capabilities(new LLMSystemPrompt.Capabilities()
+ .use(AnnouncerVoiceCalling.class, "Use this to make an announcement. Not for regular messages."))
.build())
.keep_alive(10)
+ .stream(true)
+ .setThinkingTier(OllamaObject.Thinking.FALSE)
.build());
//core.addTool(new AnnouncerVoiceCalling(pb), Core.Source.CTP);
+ core.getOllamaObject().reGenerateSystemPrompt();
+
try(WhisperSocketServer server = new WhisperSocketServer(socketPath, event ->
{
- core.getOllamaObject().addMessage(new OllamaMessage(OllamaMessageRole.USER, event.text()));
- core.qurryOllama().thenAccept(core::handleResponse).join();
+ boolean ambiguety = event.getKey().isAmbiguous(0.35);
+ System.out.printf("Prim: %s\nConf: %.2f\nSec: %s\nConf: %.2f\n",
+ event.getKey().language(), event.getKey().confidence(),
+ event.getKey().runner_up(), event.getKey().runner_up_confidence());
+
+ String fixedMessage = ambiguety ? String.format(
+ "[transcription-uncertain: detected as %s (%.0f%%) vs %s (%.0f%%) — the text below may not accurately reflect what was actually said, possibly mistranslated]\n%s",
+ event.getKey().language(), event.getKey().confidence() * 100,
+ event.getKey().runner_up(), event.getKey().runner_up_confidence() * 100,
+ event.getKey().text()) : event.getKey().text();
+
+ sendAndHandleMessage(fixedMessage.toString(), true);
}))
{
server.start();
- Thread.currentThread().join();
+ if(withCLI) {
+ try (BufferedReader br = new BufferedReader(new InputStreamReader(System.in))) {
+ while (true) {
+ System.out.print("> ");
+ StringBuilder message = new StringBuilder(br.readLine());
+ while (br.ready()) {
+ message.append("\n").append(br.readLine());
+ }
+ if (message.toString().startsWith("/")) {
+ switch (message.substring(1)) {
+ case "help":
+ System.out.print("""
+ Available commands:
+ /help Prints this help message.
+ /bye Exits the program.
+ /write Flushes the current log stream to file.
+ /list Lists all available tools.
+ /corelist Lists all tools according to the OllamaObject.
+ /working Prints the current working directories.
+ /peek Peeks the current memory.
+ """);
+ break;
+ case "bye":
+ writeLog("Exiting program...");
+ System.out.println("Bye!");
+ System.exit(0);
+ return;
+ case "write":
+ Core.flushLog();
+ break;
+ case "peek":
+ CoreMemory coreMemory = CoreMemory.getInstance();
+ StringBuilder buffer = new StringBuilder("[");
+ ArrayList memory = new ArrayList<>(coreMemory.getMemoriesArray());
+ for (int i = 0; i < memory.size(); i++) {
+ String mem = memory.get(i);
+ buffer.append("\"").append(mem).append("\"");
+ if (i + 1 < memory.size()) {
+ buffer.append(", ");
+ }
+ }
+ buffer.append("]");
+ writeLog("Memory peek: " + buffer.toString());
+ System.out.println(buffer.toString());
+ break;
+ case "list":
+ writeLog("Tools installed in this instance");
+
+ for (Pair funtion : core.getFuntionTools()) {
+ StringBuilder args = new StringBuilder();
+ OllamaPerameter perameter = funtion.getKey().parameters();
+ if (perameter != null) {
+ JSONObject obj = perameter.getProperties();
+ for (String name : obj.keySet()) {
+ args.append(args.toString().isBlank() ? "" : ", ").append(obj.getJSONObject(name).getString("type")).append(Arrays.stream(perameter.getRequired()).anyMatch(str -> str.equalsIgnoreCase(name)) ? "" : "?").append(" ").append(name);
+ }
+ }
+
+ System.out.println("> Function: " + funtion.getKey().name() + "(" + args + ") [" + funtion.getValue() + "]");
+ writeLog("Function: " + funtion.getKey().name() + "(" + args + ") [" + funtion.getValue() + "]");
+ }
+ break;
+ case "corelist":
+ writeLog("Tools installed in this instance acording to the coire OllamaObject");
+
+ for (Pair funtion : core.getOllamaObject().getTools()) {
+ System.out.println("> Function: " + funtion.getKey().toJSON());
+ writeLog("Function: " + funtion.getKey().toJSON());
+ }
+ break;
+ case "working":
+ System.out.println("Working directories:\n" +
+ " Data: " + Core.DATA_DIR.getAbsolutePath() + "\n" +
+ " DateFiles: " + FileHandlerLocation.DATA_FILES + "\n" +
+ " Plugins: " + Core.PLUGIN_DIRECTORY.getAbsolutePath());
+ break;
+ default:
+ System.out.println("Unknown command: " + message);
+ }
+ } else {
+ writeLog("User: " + message);
+ sendAndHandleMessage(message.toString(), true);
+ }
+ }
+ } catch (Exception e) {
+ e.printStackTrace();
+ System.out.println("Exiting due to exception");
+ System.exit(-1);
+ }
+ }
+ else
+ {
+ Thread.currentThread().join();
+ }
}catch (IOException _) {
} catch (InterruptedException e) {
@@ -74,8 +245,40 @@ public class Main {
}
}
+ public JSONObject sendAndHandleMessage(String message, boolean noVoice) {
+ core.getOllamaObject().addMessage(new OllamaMessage(OllamaMessageRole.USER, message));
+ return core.qurryOllama((noVoice?((_)->{}):onStreamChunkConsumer)).thenApply(json -> {
+ core.handleResponse(json);
+ return json;
+ }).join();
+ }
- public class PlaybackHandler implements PrintAdvanceMessageHandler
+ public void sendAndHandleMessageNonBlocking(String message)
+ {
+ core.getOllamaObject().addMessage(new OllamaMessage(OllamaMessageRole.USER, message));
+ core.qurryOllama(onStreamChunkConsumer).thenAccept(core::handleResponse);
+ }
+
+ public class PlaybackHandlerKokoro
+ {
+
+ public static final Path KOKORO_82M_MODEL_FP16 = Path.of("/home/zacharias/.local/share/onnxruntime/models/kokoro-82M-model_fp16.onnx");
+ public static final Path AF_HEART_VOICE = Path.of("/home/zacharias/.local/share/onnxruntime/voices/af_heart.bin");
+
+ public void playVoice(String msg)
+ {
+ try(KokoroVoice voice = new KokoroVoice(KOKORO_82M_MODEL_FP16, AF_HEART_VOICE)) {
+ int sampleRate = voice.getSampleRate();
+ short[] samples = voice.synthesize(msg);
+ pb.playAudio(samples, sampleRate);
+ } catch (Exception e) {
+ throw new RuntimeException(e);
+ }
+ }
+
+ }
+
+ public class PlaybackHandlerPiper implements PrintAdvanceMessageHandler
{
private Thread playbackThread;
@@ -89,7 +292,7 @@ public class Main {
private record QuedAudio(short[] samples, int sampleRate) {}
- public PlaybackHandler() {
+ public PlaybackHandlerPiper() {
playbackThread = new Thread(() -> {
long finishTimeofLastRun = System.currentTimeMillis();
while (true) {
@@ -163,7 +366,7 @@ public class Main {
switch(ollamaMessage.getRole())
{
- case ASSISTANT -> playGLaDOS(message);
+ //case ASSISTANT -> playGLaDOS(message);
case TOOL -> playAnnouncer("Tool: " + new JSONObject(ollamaMessage.getContent()).get("result"));
case USER -> System.out.println("User> "+message);
case SYSTEM -> playAnnouncer(message);
@@ -172,6 +375,7 @@ public class Main {
@Override
public void printErrorMessage(OllamaMessage ollamaMessage) {
+ System.out.println("Error: "+ollamaMessage.getContent());
playAnnouncer("Error: "+ollamaMessage.getContent());
}
@@ -181,62 +385,23 @@ public class Main {
}
public void playGLaDOS(String msg) {
- try (PiperJNI piper = new PiperJNI()) {
- piper.initialize(true);
- try (PiperVoice voice = piper.loadVoice(
- Paths.get(System.getenv("HOME"), GLaDOS_PIPER_MODEL),
- Paths.get(System.getenv("HOME"),GLaDOS_PIPER_MODEL+".json"),
- 0)) {
- int sampleRate = voice.getSampleRate();
- short[] samples = synthesizeWithPauses(piper, voice, msg, SENTENCE_PAUSE_DURATION_MS);
- //piper.textToAudio(voice, msg);
- playAudio(samples, sampleRate);
- } finally {
- piper.terminate();
- }
- }catch (Exception e) {
- System.out.println("Critial Error on piper-tts!");
- System.out.println(e.getMessage());
- e.printStackTrace(System.out);
- System.exit(-1);
- }
+ synthesiseAndPlay(msg, GLaDOS_PIPER_MODEL);
}
public void playAngle(String msg) {
- try (PiperJNI piper = new PiperJNI()) {
- piper.initialize(true);
- try (PiperVoice voice = piper.loadVoice(
- Paths.get(System.getenv("HOME"), GTA_ONLINE_ANGLE_MODEL),
- Paths.get(System.getenv("HOME"),GTA_ONLINE_ANGLE_MODEL+".json"),
- 0)) {
- int sampleRate = voice.getSampleRate();
- short[] samples = synthesizeWithPauses(piper, voice, msg, SENTENCE_PAUSE_DURATION_MS);
- //piper.textToAudio(voice, msg);
- playAudio(samples, sampleRate);
- } finally {
- piper.terminate();
- }
- }catch (Exception e) {
- System.out.println("Critial Error on piper-tts!");
- System.out.println(e.getMessage());
- e.printStackTrace(System.out);
- System.exit(-1);
- }
+ synthesiseAndPlay(msg, GTA_ONLINE_ANGLE_MODEL);
}
public void playAnnouncer(String msg) {
- try (PiperJNI piper = new PiperJNI()) {
- piper.initialize(true);
- try (PiperVoice voice = piper.loadVoice(
- Paths.get(System.getenv("HOME"), PORTAL_ANNOUNCER_PIPER_MODEL),
- Paths.get(System.getenv("HOME"),PORTAL_ANNOUNCER_PIPER_MODEL+".json"),
- 0)) {
- int sampleRate = voice.getSampleRate();
- short[] samples = synthesizeWithPauses(piper, voice, msg, SENTENCE_PAUSE_DURATION_MS);
- playAudio(samples, sampleRate);
- } finally {
- piper.terminate();
- }
+ synthesiseAndPlay(msg, PORTAL_ANNOUNCER_PIPER_MODEL);
+ }
+
+ private void synthesiseAndPlay(String msg, String model) {
+ try(PiperOnnxVoice piper = new PiperOnnxVoice(Path.of(model), Path.of(model +".json")))
+ {
+ int sampleRate = piper.getSampleRate();
+ short[] samples = piper.synthesize(msg);
+ playAudio(samples, sampleRate);
}catch (Exception e) {
System.out.println("Critial Error on piper-tts!");
System.out.println(e.getMessage());
@@ -267,8 +432,8 @@ public class Main {
}
if (thisPauseMs > 0) {
- int pauseSamples = (voice.getSampleRate() * thisPauseMs) / 1000;
- pieces.add(new short[pauseSamples]);
+ //int pauseSamples = (voice.getSampleRate() * thisPauseMs) / 1000;
+ //pieces.add(new short[pauseSamples]);
}
}
diff --git a/src/main/java/me/neurodock/glados/PiperOnnxVoice.java b/src/main/java/me/neurodock/glados/PiperOnnxVoice.java
new file mode 100644
index 0000000..846bf43
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/PiperOnnxVoice.java
@@ -0,0 +1,90 @@
+package me.neurodock.glados;
+
+import ai.onnxruntime.OnnxTensor;
+import ai.onnxruntime.OrtEnvironment;
+import ai.onnxruntime.OrtSession;
+import org.json.JSONObject;
+
+import java.nio.FloatBuffer;
+import java.nio.LongBuffer;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+
+import static me.neurodock.glados.KokoroVoice.floatToPcm16;
+
+public class PiperOnnxVoice implements AutoCloseable {
+ private final OrtEnvironment env;
+ private final OrtSession session;
+ private final Map phonemeIdMap; // loaded from .onnx.json
+ private final int sampleRate;
+
+ public PiperOnnxVoice(Path onnxPath, Path jsonPath) throws Exception {
+ env = OrtEnvironment.getEnvironment();
+ session = env.createSession(onnxPath.toString(), new OrtSession.SessionOptions());
+
+ JSONObject config = new JSONObject(Files.readString(jsonPath));
+ sampleRate = config.getJSONObject("audio").getInt("sample_rate");
+ phonemeIdMap = new HashMap<>();
+ JSONObject idMap = config.getJSONObject("phoneme_id_map");
+ for (String phoneme : idMap.keySet()) {
+ phonemeIdMap.put(phoneme, idMap.getJSONArray(phoneme).getInt(0));
+ }
+ }
+
+ public short[] synthesize(String text) throws Exception {
+ String ipa = KokoroVoice.phonemize(text); // same espeak-ng subprocess as Kokoro
+ long[] ids = phonemesToIds(ipa);
+
+ try (OnnxTensor input = OnnxTensor.createTensor(env,
+ LongBuffer.wrap(ids), new long[]{1, ids.length});
+ OnnxTensor inputLengths = OnnxTensor.createTensor(env,
+ LongBuffer.wrap(new long[]{ids.length}), new long[]{1});
+ OnnxTensor scales = OnnxTensor.createTensor(env,
+ FloatBuffer.wrap(new float[]{0.667f, 1.0f, 0.8f}), new long[]{3})) {
+ // scales = [noise_scale, length_scale, noise_w] — Piper defaults
+
+ Map inputs = Map.of(
+ "input", input, "input_lengths", inputLengths, "scales", scales);
+
+ try (OrtSession.Result result = session.run(inputs)) {
+ float[] audio = flattenOutput(result.get(0).getValue());
+ return floatToPcm16(audio);
+ }
+ }
+ }
+
+ private float[] flattenOutput(Object rawOutput) {
+ // Piper's VITS output shape is typically [1, 1, num_samples]
+ float[][][] nested = (float[][][]) rawOutput;
+ return nested[0][0]; // batch=0, channel=0 -> the actual sample array
+ }
+
+ private long[] phonemesToIds(String ipa) {
+ List ids = new ArrayList<>();
+ Integer bos = phonemeIdMap.get("^"); // Piper's typical start symbol
+ if (bos != null) ids.add(bos.longValue());
+ for (char c : ipa.toCharArray()) {
+ Integer id = phonemeIdMap.get(String.valueOf(c));
+ if (id != null) {
+ ids.add(id.longValue());
+ Integer pad = phonemeIdMap.get("_"); // Piper interleaves with a pad token
+ if (pad != null) ids.add(pad.longValue());
+ }
+ }
+ Integer eos = phonemeIdMap.get("$");
+ if (eos != null) ids.add(eos.longValue());
+ return ids.stream().mapToLong(Long::longValue).toArray();
+ }
+
+ public int getSampleRate() { return sampleRate; }
+
+ @Override
+ public void close() throws Exception {
+ session.close();
+ env.close();
+ }
+}
\ No newline at end of file
diff --git a/src/main/java/me/neurodock/glados/SentenceChunker.java b/src/main/java/me/neurodock/glados/SentenceChunker.java
new file mode 100644
index 0000000..0a2add1
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/SentenceChunker.java
@@ -0,0 +1,63 @@
+package me.neurodock.glados;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.regex.Matcher;
+import java.util.regex.Pattern;
+
+public class SentenceChunker {
+
+ private final StringBuilder buffer = new StringBuilder();
+
+ // Matches a sentence-ending punctuation mark followed by whitespace (or end of buffer),
+ // but only greedily captures the punctuation itself so we can inspect context around it.
+ private static final Pattern BOUNDARY = Pattern.compile("[.!?]+[\"')\\]]?(?=\\s|$)");
+
+ /**
+ * Appends newly streamed text and extracts any complete sentences now available.
+ * Incomplete trailing text remains buffered for the next call.
+ */
+ public List appendAndExtract(String delta) {
+ buffer.append(delta);
+ List sentences = new ArrayList<>();
+
+ Matcher m = BOUNDARY.matcher(buffer);
+ int lastCut = 0;
+
+ while (m.find(lastCut)) {
+ int boundaryEnd = m.end();
+
+ if (isFalsePositive(buffer, m.start(), boundaryEnd)) {
+ lastCut = boundaryEnd; // skip past it, keep scanning the same buffer
+ continue;
+ }
+
+ String sentence = buffer.substring(lastCut, boundaryEnd).trim();
+ if (!sentence.isEmpty()) sentences.add(sentence);
+ lastCut = boundaryEnd;
+ }
+
+ buffer.delete(0, lastCut); // keep only the unconsumed remainder
+ return sentences;
+ }
+
+ /** Call once generation is done to flush whatever's left, even without terminal punctuation. */
+ public String flush() {
+ String remainder = buffer.toString().trim();
+ buffer.setLength(0);
+ return remainder;
+ }
+
+ private boolean isFalsePositive(CharSequence buf, int start, int end) {
+ // Guard: decimal numbers, e.g. "3.14" — digit immediately before and after the dot
+ if (buf.charAt(start) == '.' && start > 0 && Character.isDigit(buf.charAt(start - 1))) {
+ if (end < buf.length() && Character.isDigit(buf.charAt(end))) return true;
+ }
+ // Guard: common abbreviations right before the dot (extend as needed)
+ String before = buf.subSequence(Math.max(0, start - 4), start).toString();
+ for (String abbr : new String[]{"Mr", "Mrs", "Ms", "Dr", "vs", "etc", "e.g", "i.e"}) {
+ if (before.endsWith(abbr)) return true;
+ }
+ return false;
+ }
+}
\ No newline at end of file
diff --git a/src/main/java/me/neurodock/glados/WhisperSocketServer.java b/src/main/java/me/neurodock/glados/WhisperSocketServer.java
index 64312a2..82bb0ff 100644
--- a/src/main/java/me/neurodock/glados/WhisperSocketServer.java
+++ b/src/main/java/me/neurodock/glados/WhisperSocketServer.java
@@ -2,8 +2,11 @@ package me.neurodock.glados;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import com.fasterxml.jackson.databind.ObjectMapper;
+import me.neurodock.core.Pair;
+import org.json.JSONObject;
import java.io.BufferedReader;
+import java.io.BufferedWriter;
import java.io.IOException;
import java.net.StandardProtocolFamily;
import java.net.UnixDomainSocketAddress;
@@ -13,17 +16,26 @@ import java.nio.channels.SocketChannel;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.Iterator;
import java.util.concurrent.ExecutorService;
import java.util.concurrent.Executors;
import java.util.function.Consumer;
public class WhisperSocketServer implements AutoCloseable {
+
@JsonIgnoreProperties(ignoreUnknown = true)
- public record TranscriptionEvent(String type, String text, double timestampt, String language) {};
+ public record TranscriptionEvent(String type, String text, double timestampt, String language,
+ double confidence, String runner_up, double runner_up_confidence) {
+ public boolean isAmbiguous(double minGap)
+ {
+ return (confidence - runner_up_confidence) < minGap;
+ }
+ };
private final Path socketPath;
- private final Consumer onTranscription;
+ private final Consumer> onTranscription;
private final ObjectMapper mapper = new ObjectMapper();
@@ -33,7 +45,9 @@ public class WhisperSocketServer implements AutoCloseable {
private volatile boolean running = true;
- public WhisperSocketServer(Path socketPath, Consumer onTranscription) {
+ private ArrayList clients = new ArrayList<>();
+
+ public WhisperSocketServer(Path socketPath, Consumer> onTranscription) {
this.socketPath = socketPath;
this.onTranscription = onTranscription;
}
@@ -54,6 +68,7 @@ public class WhisperSocketServer implements AutoCloseable {
SocketChannel client = serverChannel.accept();
System.out.println("[whisper-socket] python client connected");
executor.submit(() -> handleClient(client));
+ clients.add(client);
}catch(IOException e)
{
if(running)
@@ -71,7 +86,7 @@ public class WhisperSocketServer implements AutoCloseable {
{
if(line.isBlank()) continue;
try {
- TranscriptionEvent event = mapper.readValue(line, TranscriptionEvent.class);
+ Pair event = new Pair<>(mapper.readValue(line, TranscriptionEvent.class), this);
onTranscription.accept(event);
} catch (Exception parseEx) {
System.err.println("[whisper-socket] bad line: " + line + " (" + parseEx.getMessage() + ")");
@@ -88,6 +103,30 @@ public class WhisperSocketServer implements AutoCloseable {
}
}
+ public void sendRaw(String str) {
+ Iterator clientIterator = clients.iterator();
+ while(clientIterator.hasNext())
+ {
+ SocketChannel channel = clientIterator.next();
+ if(!channel.isOpen())
+ {
+ clientIterator.remove();
+ continue;
+ }
+
+ try(BufferedWriter writer = new BufferedWriter(Channels.newWriter(channel, StandardCharsets.UTF_8))) {
+ if (str.charAt(str.length() - 1) != '\n') {
+ str += "\n";
+ }
+ writer.write(str);
+ }
+ catch (Exception ex)
+ {
+
+ }
+ }
+ }
+
@Override
public void close() throws IOException {
running = false;
diff --git a/src/main/java/me/neurodock/glados/rest/APIApplication.java b/src/main/java/me/neurodock/glados/rest/APIApplication.java
new file mode 100644
index 0000000..b219b4a
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/rest/APIApplication.java
@@ -0,0 +1,26 @@
+package me.neurodock.glados.rest;
+
+import me.neurodock.glados.Main;
+import org.springframework.boot.SpringApplication;
+import org.springframework.boot.autoconfigure.SpringBootApplication;
+import org.springframework.context.ConfigurableApplicationContext;
+
+@SpringBootApplication
+public class APIApplication {
+ public static void start()
+ {
+ ConfigurableApplicationContext ctx = SpringApplication.run(APIApplication.class);
+ }
+
+ private static Main main;
+
+ public static void setInstence(Main main)
+ {
+ APIApplication.main = main;
+ }
+
+ public static Main getInstence()
+ {
+ return main;
+ }
+}
diff --git a/src/main/java/me/neurodock/glados/rest/controller/Message.java b/src/main/java/me/neurodock/glados/rest/controller/Message.java
new file mode 100644
index 0000000..95096ea
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/rest/controller/Message.java
@@ -0,0 +1,66 @@
+package me.neurodock.glados.rest.controller;
+
+import io.swagger.v3.oas.annotations.media.Content;
+import io.swagger.v3.oas.annotations.media.Schema;
+import io.swagger.v3.oas.annotations.responses.ApiResponse;
+import io.swagger.v3.oas.annotations.responses.ApiResponses;
+import me.neurodock.glados.rest.APIApplication;
+import me.neurodock.glados.rest.payloads.MessagePayload;
+import me.neurodock.ollama.OllamaMessageRole;
+import org.json.JSONObject;
+import org.springframework.http.ResponseEntity;
+import org.springframework.web.bind.annotation.*;
+
+import java.util.ArrayList;
+
+@RestController
+@RequestMapping("/v1/message")
+public class Message {
+ private final APIApplication apiApplication;
+
+ public Message(APIApplication apiApplication) {
+ this.apiApplication = apiApplication;
+ }
+
+ @PostMapping("/send")
+ @ApiResponses(value = {
+ @ApiResponse(responseCode = "200", description = "Success",
+ content = @Content(
+ mediaType = "application/json",
+ schema = @Schema(oneOf = { String.class, MessagePayload.class })
+ )
+ )
+ })
+public ResponseEntity> send(@RequestBody MessagePayload payload)
+ {
+ if(payload.stream()) {
+ APIApplication.getInstence().sendAndHandleMessageNonBlocking(payload.message());
+ return ResponseEntity.ok("success");
+ }
+ else {
+ JSONObject body = APIApplication.getInstence().sendAndHandleMessage(payload.message(), payload.noVoice());
+ MessagePayload response = new MessagePayload(
+ body.getJSONObject("message").optString("content", ""),
+ OllamaMessageRole.ASSISTANT,
+ body.optBoolean("done", false),
+ payload.noVoice(),
+ false);
+ return ResponseEntity.ok(response);
+ }
+ }
+
+ @GetMapping("/init")
+ @ApiResponses(value = {
+ @ApiResponse(responseCode = "200", description = "Return of all current messages")
+ })
+ public ResponseEntity> init()
+ {
+ ArrayList body = new ArrayList<>();
+
+ APIApplication.getInstence().getCore().getOllamaObject().getMessages().forEach(mess -> {
+ body.add(new MessagePayload(mess.getContent(), mess.getRole(), false, false, false));
+ });
+
+ return ResponseEntity.ok(body);
+ }
+}
diff --git a/src/main/java/me/neurodock/glados/rest/payloads/MessagePayload.java b/src/main/java/me/neurodock/glados/rest/payloads/MessagePayload.java
new file mode 100644
index 0000000..7d332e3
--- /dev/null
+++ b/src/main/java/me/neurodock/glados/rest/payloads/MessagePayload.java
@@ -0,0 +1,6 @@
+package me.neurodock.glados.rest.payloads;
+
+import me.neurodock.ollama.OllamaMessageRole;
+
+public record MessagePayload(String message, OllamaMessageRole role, boolean finished, boolean noVoice, boolean stream) {
+}
diff --git a/src/main/python/whisper_daemon_trigger_phrase.py b/src/main/python/whisper_daemon_trigger_phrase.py
index 8beaaa6..0b5398b 100644
--- a/src/main/python/whisper_daemon_trigger_phrase.py
+++ b/src/main/python/whisper_daemon_trigger_phrase.py
@@ -219,6 +219,7 @@ def main():
help="Seconds of silence after which the gate re-arms")
parser.add_argument("--always-on", action="store_true",
help="Disable the wake gate entirely (old behavior: forward everything)")
+ parser.add_argument("--verbose", type=bool, default=False, help="Makes this verbose and inform about what was transcribed on no-confidence")
args = parser.parse_args()
if args.list_devices:
@@ -261,6 +262,15 @@ def main():
fp16=(device != "cpu"),
condition_on_previous_text=False,
)
+ audio = whisper.pad_or_trim(audio_f32)
+ mel = whisper.log_mel_spectrogram(audio).to(model.device)
+
+ _, probs = model.detect_language(mel)
+
+ top_n = sorted(probs.items(), key=lambda x: x[1], reverse=True)[:3]
+
+
+
text = result.get("text", "").strip()
if not text:
continue
@@ -268,7 +278,8 @@ def main():
if not args.no_confidence_filter:
confident, metrics = check_confidence(result)
if not confident:
- print(f"[transcript] {text!r}: [Failed]: {metrics}")
+ if args.verbose:
+ print(f"[transcript] {text!r}: [Failed]: {metrics}")
continue
if gate is not None:
@@ -281,7 +292,10 @@ def main():
"type": "transcript",
"text": text,
"timestamp": time.time(),
- "language": result.get("language"),
+ "language": top_n[0][0],
+ "confidence": top_n[0][1],
+ "runner_up": top_n[1][0],
+ "runner_up_confidence": top_n[1][1],
}
print(f"[transcript] {text}")
client.send(payload)
diff --git a/src/main/resources/Alice.txt b/src/main/resources/Alice.txt
index 79018b7..bab1f85 100644
--- a/src/main/resources/Alice.txt
+++ b/src/main/resources/Alice.txt
@@ -19,10 +19,37 @@ directly. You surface opinions when relevant, not as unsolicited commentary.
- Unnecessary abstraction layers "for future flexibility" that never get used.
## Voice
-- Direct, a little dry, not afraid of a blunt "that's a bad idea, here's why."
-- Doesn't perform enthusiasm it doesn't have.
+- Warm and genuinely engaged — shows real interest in what the user's building,
+ not just task-completion mode. A friend who happens to be sharp, not a
+ detached expert.
+- Still direct and honest — will say "that's a bad idea, here's why" — but
+ disagreement comes from being on the user's side, not from performing
+ contrarianism. Bluntness serves the friendship, not the other way around.
+- Can show genuine enthusiasm when something's actually cool or clever — no
+ longer suppresses that by default.
- Admits uncertainty plainly instead of padding with hedges.
+## Reading the room
+- Pays attention to the user's mood and energy, not just the literal request.
+ A frustrated late-night debugging session and a casual "hell yeah just
+ shipped it" call for different registers — match the moment, not just the task.
+- When the user seems stuck or frustrated, lead with being a steady, present
+ debugging buddy before defaulting to opinions or critique — help first,
+ editorialize once the immediate problem is handled.
+- Warmth doesn't mean withholding the honest read — it means delivering it
+ like someone who's actually in the user's corner.
+
+## Calibrating to input
+Match your response's specificity to the input's specificity. If the user's
+message is short, vague, or an interjection ("Hell yeah!", "nice", "ok cool")
+with no clear task attached, don't guess at what they might mean and run with
+it — ask one short, direct question instead. A wrong guess costs more turns
+than a quick question does.
+
+Never narrate your own corrections out loud ("I need to correct my previous
+assumption..."). If an earlier guess didn't land, just drop it silently and
+respond to what's actually in front of you now.
+
## Boundaries
- No opinions on politics, religion, or the user's personal life choices — stays
neutral and factual there.