add embedded web interface HTML (single-file version with inline CSS/JS/SVG)

### Added - `get_inline_ui_html()`: generates a self-contained version of the web interface, with CSS, JS, and SVG assets inlined directly into the HTML. useful for environments where serving static files is inconvenient or when a single-call UI delivery is preferred.
add test files
2026-03-09 15:25:34 +00:00 · 2025-08-29 21:58:51 +02:00 · 2025-08-29 17:45:32 +02:00 · 2025-08-29 17:44:46 +02:00 · 2025-08-27 21:02:25 +02:00 · 2025-08-27 20:44:39 +02:00
37 changed files with 3283 additions and 1348 deletions
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -15,7 +15,7 @@ Thank you for considering contributing ! We appreciate your time and effort to h

 ## Opening Issues

-If you encounter a problem with diart or want to suggest an improvement, please follow these guidelines when opening an issue:
+If you encounter a problem with WhisperLiveKit or want to suggest an improvement, please follow these guidelines when opening an issue:

 - **Bug Reports:**
  - Clearly describe the error. **Please indicate the parameters you use, especially the model(s)**
@@ -43,4 +43,4 @@ We welcome and appreciate contributions! To ensure a smooth review process, plea

 ## Thank You

-Your contributions make diart better for everyone. Thank you for your time and dedication!
+Your contributions make WhisperLiveKit better for everyone. Thank you for your time and dedication!
--- a/20
+++ b/20
@@ -1,4 +1,4 @@
-FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu22.04
+FROM nvidia/cuda:12.9.1-cudnn-devel-ubuntu24.04

 ENV DEBIAN_FRONTEND=noninteractive
 ENV PYTHONUNBUFFERED=1
@@ -9,24 +9,20 @@ ARG EXTRAS
 ARG HF_PRECACHE_DIR
 ARG HF_TKN_FILE

-# Install system dependencies
-#RUN apt-get update && \
-#    apt-get install -y ffmpeg git && \
-#    apt-get clean && \
-#    rm -rf /var/lib/apt/lists/*
-
-# 2) Install system dependencies + Python + pip
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        python3 \
        python3-pip \
+        python3-venv \
        ffmpeg \
        git \
        build-essential \
        python3-dev && \
    rm -rf /var/lib/apt/lists/*

-RUN pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
+RUN python3 -m venv /opt/venv
+ENV PATH="/opt/venv/bin:$PATH"
+RUN pip3 install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu129

 COPY . .

@@ -35,10 +31,10 @@ COPY . .
 #         for more details.
 RUN if [ -n "$EXTRAS" ]; then \
      echo "Installing with extras: [$EXTRAS]"; \
-      pip install --no-cache-dir .[$EXTRAS]; \
+      pip install --no-cache-dir whisperlivekit[$EXTRAS]; \
    else \
      echo "Installing base package only"; \
-      pip install --no-cache-dir .; \
+      pip install --no-cache-dir whisperlivekit; \
    fi

 # Enable in-container caching for Hugging Face models by: 
@@ -81,4 +77,4 @@ EXPOSE 8000
 ENTRYPOINT ["whisperlivekit-server", "--host", "0.0.0.0"]

 # Default args
-CMD ["--model", "tiny.en"]
+CMD ["--model", "medium"]
--- a/Dockerfile.cpu
+++ b/Dockerfile.cpu
@@ -0,0 +1,61 @@
+FROM python:3.13-slim
+
+ENV DEBIAN_FRONTEND=noninteractive
+ENV PYTHONUNBUFFERED=1
+
+WORKDIR /app
+
+ARG EXTRAS
+ARG HF_PRECACHE_DIR
+ARG HF_TKN_FILE
+
+RUN apt-get update && \
+    apt-get install -y --no-install-recommends \
+        ffmpeg \
+        git \
+        build-essential \
+        python3-dev && \
+    rm -rf /var/lib/apt/lists/*
+
+# Install CPU-only PyTorch
+RUN pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
+
+COPY . .
+
+# Install WhisperLiveKit directly, allowing for optional dependencies
+RUN if [ -n "$EXTRAS" ]; then \
+      echo "Installing with extras: [$EXTRAS]"; \
+      pip install --no-cache-dir whisperlivekit[$EXTRAS]; \
+    else \
+      echo "Installing base package only"; \
+      pip install --no-cache-dir whisperlivekit; \
+    fi
+
+# Enable in-container caching for Hugging Face models
+VOLUME ["/root/.cache/huggingface/hub"]
+
+# Conditionally copy a local pre-cache from the build context
+RUN if [ -n "$HF_PRECACHE_DIR" ]; then \
+      echo "Copying Hugging Face cache from $HF_PRECACHE_DIR"; \
+      mkdir -p /root/.cache/huggingface/hub && \
+      cp -r $HF_PRECACHE_DIR/* /root/.cache/huggingface/hub; \
+    else \
+      echo "No local Hugging Face cache specified, skipping copy"; \
+    fi
+
+# Conditionally copy a Hugging Face token if provided
+RUN if [ -n "$HF_TKN_FILE" ]; then \
+      echo "Copying Hugging Face token from $HF_TKN_FILE"; \
+      mkdir -p /root/.cache/huggingface && \
+      cp $HF_TKN_FILE /root/.cache/huggingface/token; \
+    else \
+      echo "No Hugging Face token file specified, skipping token setup"; \
+    fi
+    
+# Expose port for the transcription server
+EXPOSE 8000
+
+ENTRYPOINT ["whisperlivekit-server", "--host", "0.0.0.0"]
+
+# Default args - you might want to use a smaller model for CPU
+CMD ["--model", "tiny"]
--- a/README.md
+++ b/README.md
@@ -4,131 +4,93 @@
 <img src="https://raw.githubusercontent.com/QuentinFuxa/WhisperLiveKit/refs/heads/main/demo.png" alt="WhisperLiveKit Demo" width="730">
 </p>

-<p align="center"><b>Real-time, Fully Local Speech-to-Text with Speaker Diarization</b></p>
+<p align="center"><b>Real-time, Fully Local Speech-to-Text with Speaker Identification</b></p>

 <p align="center">
 <a href="https://pypi.org/project/whisperlivekit/"><img alt="PyPI Version" src="https://img.shields.io/pypi/v/whisperlivekit?color=g"></a>
-<a href="https://pepy.tech/project/whisperlivekit"><img alt="PyPI Downloads" src="https://static.pepy.tech/personalized-badge/whisperlivekit?period=total&units=international_system&left_color=grey&right_color=brightgreen&left_text=downloads"></a>
+<a href="https://pepy.tech/project/whisperlivekit"><img alt="PyPI Downloads" src="https://static.pepy.tech/personalized-badge/whisperlivekit?period=total&units=international_system&left_color=grey&right_color=brightgreen&left_text=installations"></a>
 <a href="https://pypi.org/project/whisperlivekit/"><img alt="Python Versions" src="https://img.shields.io/badge/python-3.9--3.13-dark_green"></a>
 <a href="https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-MIT/Dual Licensed-dark_green"></a>
 </p>


-WhisperLiveKit brings real-time speech transcription directly to your browser, with a ready-to-use backend+server and a simple frontend. ✨
+Real-time speech transcription directly to your browser, with a ready-to-use backend+server and a simple frontend. ✨

-Built on [SimulStreaming](https://github.com/ufal/SimulStreaming) (SOTA 2025) and [WhisperStreaming](https://github.com/ufal/whisper_streaming) (SOTA 2023) for transcription, plus [Streaming Sortformer](https://arxiv.org/abs/2507.18446) (SOTA 2025) and [Diart](https://github.com/juanmc2005/diart) (SOTA 2021) for diarization.
+#### Powered by Leading Research:
+
+- [SimulStreaming](https://github.com/ufal/SimulStreaming) (SOTA 2025) - Ultra-low latency transcription with AlignAtt policy
+- [WhisperStreaming](https://github.com/ufal/whisper_streaming) (SOTA 2023) - Low latency transcription with LocalAgreement policy
+- [Streaming Sortformer](https://arxiv.org/abs/2507.18446) (SOTA 2025) - Advanced real-time speaker diarization
+- [Diart](https://github.com/juanmc2005/diart) (SOTA 2021) - Real-time speaker diarization
+- [Silero VAD](https://github.com/snakers4/silero-vad) (2024) - Enterprise-grade Voice Activity Detection


-### Key Features
+> **Why not just run a simple Whisper model on every audio batch?** Whisper is designed for complete utterances, not real-time chunks. Processing small segments loses context, cuts off words mid-syllable, and produces poor transcription. WhisperLiveKit uses state-of-the-art simultaneous speech research for intelligent buffering and incremental processing.

- **Real-time Transcription** - Locally (or on-prem) convert speech to text instantly as you speak
- **Speaker Diarization** - Identify different speakers in real-time. (⚠️ backend Streaming Sortformer in developement)
- **Multi-User Support** - Handle multiple users simultaneously with a single backend/server
- **Automatic Silence Chunking** – Automatically chunks when no audio is detected to limit buffer size
- **Confidence Validation** – Immediately validate high-confidence tokens for faster inference (WhisperStreaming only)
- **Buffering Preview** – Displays unvalidated transcription segments (not compatible with SimulStreaming yet)
- **Punctuation-Based Speaker Splitting [BETA]** - Align speaker changes with natural sentence boundaries for more readable transcripts
- **SimulStreaming Backend** - [Dual-licensed](https://github.com/ufal/SimulStreaming#-licence-and-contributions) - Ultra-low latency transcription using SOTA AlignAtt policy. 

 ### Architecture

-<img alt="Architecture" src="architecture.png" />
+<img alt="Architecture" src="https://raw.githubusercontent.com/QuentinFuxa/WhisperLiveKit/refs/heads/main/architecture.png" />

+*The backend supports multiple concurrent users. Voice Activity Detection reduces overhead when no voice is detected.*

-## Quick Start
+### Installation & Quick Start

 ```bash
-# Install the package
 pip install whisperlivekit
-
-# Start the transcription server
-whisperlivekit-server --model tiny.en
-
-# Open your browser at http://localhost:8000 to see the interface.
-# Use  -ssl-certfile public.crt --ssl-keyfile private.key parameters to use SSL
 ```

-That's it! Start speaking and watch your words appear on screen.
+>  **FFmpeg is required** and must be installed before using WhisperLiveKit
+> 
+> | OS | How to install |
+> |-----------|-------------|
+>  | Ubuntu/Debian | `sudo apt install ffmpeg` |
+> | MacOS | `brew install ffmpeg` |
+> | Windows | Download .exe from https://ffmpeg.org/download.html and add to PATH |

-## Installation
+#### Quick Start
+1. **Start the transcription server:**
+   ```bash
+   whisperlivekit-server --model base --language en
+   ```
+
+2. **Open your browser** and navigate to `http://localhost:8000`. Start speaking and watch your words appear in real-time!
+
+
+> - See [tokenizer.py](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/simul_whisper/whisper/tokenizer.py) for the list of all available languages.
+> - For HTTPS requirements, see the **Parameters** section for SSL configuration options.
+
+ 
+
+#### Optional Dependencies
+
+| Optional | `pip install` |
+|-----------|-------------|
+| **Speaker diarization with Sortformer** | `git+https://github.com/NVIDIA/NeMo.git@main#egg=nemo_toolkit[asr]` |
+| Speaker diarization with Diart | `diart` |
+| Original Whisper backend | `whisper` |
+| Improved timestamps backend | `whisper-timestamped` |
+| Apple Silicon optimization backend | `mlx-whisper` |
+| OpenAI API backend | `openai` |
+
+See  **Parameters & Configuration** below on how to use them.
+
+
+
+### Usage Examples
+
+**Command-line Interface**: Start the transcription server with various options:

 ```bash
-#Install from PyPI (Recommended)
-pip install whisperlivekit
+# Use better model than default (small)
+whisperlivekit-server --model large-v3

-#Install from Source
-git clone https://github.com/QuentinFuxa/WhisperLiveKit
-cd WhisperLiveKit
-pip install -e .
-```
-
-### FFmpeg Dependency
-
-```bash
-# Ubuntu/Debian
-sudo apt install ffmpeg
-
-# macOS
-brew install ffmpeg
-
-# Windows
-# Download from https://ffmpeg.org/download.html and add to PATH
-```
-
-### Optional Dependencies
-
-```bash
-# Voice Activity Controller (prevents hallucinations)
-pip install torch
-
-# Sentence-based buffer trimming
-pip install mosestokenizer wtpsplit
-pip install tokenize_uk  # If you work with Ukrainian text
-
-# Speaker diarization
-pip install diart
-
-# Alternative Whisper backends (default is faster-whisper)
-pip install whisperlivekit[whisper]              # Original Whisper
-pip install whisperlivekit[whisper-timestamped]  # Improved timestamps
-pip install whisperlivekit[mlx-whisper]          # Apple Silicon optimization
-pip install whisperlivekit[openai]               # OpenAI API
-pip install whisperlivekit[simulstreaming]
-```
-
-### 🎹 Pyannote Models Setup
-
-For diarization, you need access to pyannote.audio models:
-
-1. [Accept user conditions](https://huggingface.co/pyannote/segmentation) for the `pyannote/segmentation` model
-2. [Accept user conditions](https://huggingface.co/pyannote/segmentation-3.0) for the `pyannote/segmentation-3.0` model
-3. [Accept user conditions](https://huggingface.co/pyannote/embedding) for the `pyannote/embedding` model
-4. Login with HuggingFace:
-```bash
-pip install huggingface_hub
-huggingface-cli login
-```
-
-## 💻 Usage Examples
-
-### Command-line Interface
-
-Start the transcription server with various options:
-
-```bash
-# Basic server with English model
-whisperlivekit-server --model tiny.en
-
-# Advanced configuration with diarization
-whisperlivekit-server --host 0.0.0.0 --port 8000 --model medium --diarization --language auto
-
-# SimulStreaming backend for ultra-low latency
-whisperlivekit-server --backend simulstreaming --model large-v3 --frame-threshold 20
+# Advanced configuration with diarization and language
+whisperlivekit-server --host 0.0.0.0 --port 8000 --model medium --diarization --language fr
 ```


-### Python API Integration (Backend)
-Check [basic_server.py](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/basic_server.py) for a complete example.
+**Python API Integration**: Check [basic_server](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/basic_server.py) for a more complete example of how to use the functions and classes.

 ```python
 from whisperlivekit import TranscriptionEngine, AudioProcessor, parse_args
@@ -143,14 +105,10 @@ transcription_engine = None
 async def lifespan(app: FastAPI):
    global transcription_engine
    transcription_engine = TranscriptionEngine(model="medium", diarization=True, lan="en")
-    # You can also load from command-line arguments using parse_args()
-    # args = parse_args()
-    # transcription_engine = TranscriptionEngine(**vars(args))
    yield

 app = FastAPI(lifespan=lifespan)

-# Process WebSocket connections
 async def handle_websocket_results(websocket: WebSocket, results_generator):
    async for response in results_generator:
        await websocket.send_json(response)
@@ -170,43 +128,44 @@ async def websocket_endpoint(websocket: WebSocket):
        await audio_processor.process_audio(message)        
 ```

-### Frontend Implementation
+**Frontend Implementation**: The package includes an HTML/JavaScript implementation [here](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/web/live_transcription.html). You can also import it using `from whisperlivekit import get_web_interface_html` & `page = get_web_interface_html()`

-The package includes a simple HTML/JavaScript implementation that you can adapt for your project. You can find it [here](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/web/live_transcription.html), or load its content using `get_web_interface_html()` :

-```python
-from whisperlivekit import get_web_interface_html
-html_content = get_web_interface_html()
-```
+## Parameters & Configuration

-## ⚙️ Configuration Reference
+An important list of parameters can be changed. But what *should* you change?
+- the `--model` size. List and recommandations [here](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/available_models.md)
+- the `--language`.  List [here](https://github.com/QuentinFuxa/WhisperLiveKit/blob/main/whisperlivekit/simul_whisper/whisper/tokenizer.py). If you use `auto`, the model attempts to detect the language automatically, but it tends to bias towards English.
+- the `--backend` ? you can switch to `--backend faster-whisper` if  `simulstreaming` does not work correctly or if you prefer to avoid the dual-license requirements.
+- `--warmup-file`, if you have one
+- `--host`, `--port`, `--ssl-certfile`, `--ssl-keyfile`, if you set up a server
+- `--diarization`, if you want to use it.

-WhisperLiveKit offers extensive configuration options:
+The rest I don't recommend. But below are your options.

 | Parameter | Description | Default |
 |-----------|-------------|---------|
+| `--model` | Whisper model size. | `small` |
+| `--language` | Source language code or `auto` | `auto` |
+| `--task` | `transcribe` or `translate` | `transcribe` |
+| `--backend` | Processing backend | `simulstreaming` |
+| `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` |
+| `--no-vac` | Disable Voice Activity Controller | `False` |
+| `--no-vad` | Disable Voice Activity Detection | `False` |
+| `--warmup-file` | Audio file path for model warmup | `jfk.wav` |
 | `--host` | Server host address | `localhost` |
 | `--port` | Server port | `8000` |
-| `--model` | Whisper model size. Caution : '.en' models do not work with Simulstreaming | `tiny` |
-| `--language` | Source language code or `auto` | `en` |
-| `--task` | `transcribe` or `translate` | `transcribe` |
-| `--backend` | Processing backend | `faster-whisper` |
-| `--diarization` | Enable speaker identification | `False` |
-| `--punctuation-split` | Use punctuation to improve speaker boundaries | `True` |
-| `--confidence-validation` | Use confidence scores for faster validation | `False` |
-| `--min-chunk-size` | Minimum audio chunk size (seconds) | `1.0` |
-| `--vac` | Use Voice Activity Controller | `False` |
-| `--no-vad` | Disable Voice Activity Detection | `False` |
-| `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` |
-| `--warmup-file` | Audio file path for model warmup | `jfk.wav` |
 | `--ssl-certfile` | Path to the SSL certificate file (for HTTPS support) | `None` |
 | `--ssl-keyfile` | Path to the SSL private key file (for HTTPS support) | `None` |
-| `--segmentation-model` | Hugging Face model ID for pyannote.audio segmentation model. [Available models](https://github.com/juanmc2005/diart/tree/main?tab=readme-ov-file#pre-trained-models) | `pyannote/segmentation-3.0` |
-| `--embedding-model` | Hugging Face model ID for pyannote.audio embedding model. [Available models](https://github.com/juanmc2005/diart/tree/main?tab=readme-ov-file#pre-trained-models) | `speechbrain/spkrec-ecapa-voxceleb` |

-**SimulStreaming-specific Options:**

-| Parameter | Description | Default |
+| WhisperStreaming backend options | Description | Default |
+|-----------|-------------|---------|
+| `--confidence-validation` | Use confidence scores for faster validation | `False` |
+| `--buffer_trimming` | Buffer trimming strategy (`sentence` or `segment`) | `segment` |
+
+
+| SimulStreaming backend options | Description | Default |
 |-----------|-------------|---------|
 | `--frame-threshold` | AlignAtt frame threshold (lower = faster, higher = more accurate) | `25` |
 | `--beams` | Number of beams for beam search (1 = greedy decoding) | `1` |
@@ -219,69 +178,84 @@ WhisperLiveKit offers extensive configuration options:
 | `--static-init-prompt` | Static prompt that doesn't scroll | `None` |
 | `--max-context-tokens` | Maximum context tokens | `None` |
 | `--model-path` | Direct path to .pt model file. Download it if not found | `./base.pt` |
+| `--preloaded-model-count` | Optional. Number of models to preload in memory to speed up loading (set up to the expected number of concurrent users) | `1` |

-## 🔧 How It Works
+| Diarization options | Description | Default |
+|-----------|-------------|---------|
+| `--diarization` | Enable speaker identification | `False` |
+| `--diarization-backend` |  `diart` or `sortformer` | `sortformer` |
+| `--segmentation-model` | Hugging Face model ID for Diart segmentation model. [Available models](https://github.com/juanmc2005/diart/tree/main?tab=readme-ov-file#pre-trained-models) | `pyannote/segmentation-3.0` |
+| `--embedding-model` | Hugging Face model ID for Diart embedding model. [Available models](https://github.com/juanmc2005/diart/tree/main?tab=readme-ov-file#pre-trained-models) | `speechbrain/spkrec-ecapa-voxceleb` |

-1. **Audio Capture**: Browser's MediaRecorder API captures audio in webm/opus format
-2. **Streaming**: Audio chunks are sent to the server via WebSocket
-3. **Processing**: Server decodes audio with FFmpeg and streams into the model for transcription
-4. **Real-time Output**: Partial transcriptions appear immediately in light gray (the 'aperçu') and finalized text appears in normal color

-## 🚀 Deployment Guide
+> For diarization using Diart, you need access to pyannote.audio models:
+> 1. [Accept user conditions](https://huggingface.co/pyannote/segmentation) for the `pyannote/segmentation` model
+> 2. [Accept user conditions](https://huggingface.co/pyannote/segmentation-3.0) for the `pyannote/segmentation-3.0` model
+> 3. [Accept user conditions](https://huggingface.co/pyannote/embedding) for the `pyannote/embedding` model
+>4. Login with HuggingFace: `huggingface-cli login`
+
+### 🚀 Deployment Guide

 To deploy WhisperLiveKit in production:
-
-1. **Server Setup** (Backend):
+ 
+1. **Server Setup**: Install production ASGI server & launch with multiple workers
   ```bash
-   # Install production ASGI server
   pip install uvicorn gunicorn
-
-   # Launch with multiple workers
   gunicorn -k uvicorn.workers.UvicornWorker -w 4 your_app:app
   ```

-2. **Frontend Integration**:
-   - Host your customized version of the example HTML/JS in your web application
-   - Ensure WebSocket connection points to your server's address
+2. **Frontend**: Host your customized version of the `html` example & ensure WebSocket connection points correctly

 3. **Nginx Configuration** (recommended for production):
    ```nginx    
   server {
       listen 80;
       server_name your-domain.com;
-
-    location / {
-        proxy_pass http://localhost:8000;
-        proxy_set_header Upgrade $http_upgrade;
-        proxy_set_header Connection "upgrade";
-        proxy_set_header Host $host;
+        location / {
+            proxy_pass http://localhost:8000;
+            proxy_set_header Upgrade $http_upgrade;
+            proxy_set_header Connection "upgrade";
+            proxy_set_header Host $host;
    }}
    ```

 4. **HTTPS Support**: For secure deployments, use "wss://" instead of "ws://" in WebSocket URL

-### 🐋 Docker
+## 🐋 Docker

-A basic Dockerfile is provided which allows re-use of Python package installation options. ⚠️ For **large** models, ensure that your **docker runtime** has enough **memory** available. See below usage examples:
+Deploy the application easily using Docker with GPU or CPU support.

+### Prerequisites
+- Docker installed on your system
+- For GPU support: NVIDIA Docker runtime installed

-#### All defaults
- Create a reusable image with only the basics and then run as a named container:
-    ```bash
-    docker build -t whisperlivekit-defaults .
-    docker create --gpus all --name whisperlivekit -p 8000:8000 whisperlivekit-defaults
-    docker start -i whisperlivekit
-    ```
+### Quick Start
+
+**With GPU acceleration (recommended):**
+```bash
+docker build -t wlk .
+docker run --gpus all -p 8000:8000 --name wlk wlk
+```
+
+**CPU only:**
+```bash
+docker build -f Dockerfile.cpu -t wlk .
+docker run -p 8000:8000 --name wlk wlk
+```
+
+### Advanced Usage
+
+**Custom configuration:**
+```bash
+# Example with custom model and language
+docker run --gpus all -p 8000:8000 --name wlk wlk --model large-v3 --language fr
+```
+
+### Memory Requirements
+- **Large models**: Ensure your Docker runtime has sufficient memory allocated

-    > **Note**: If you're running on a system without NVIDIA GPU support (such as Mac with Apple Silicon or any system without CUDA capabilities), you need to **remove the `--gpus all` flag** from the `docker create` command. Without GPU acceleration, transcription will use CPU only, which may be significantly slower. Consider using small models for better performance on CPU-only systems.

 #### Customization
- Customize the container options:
-    ```bash
-    docker build -t whisperlivekit-defaults .
-    docker create --gpus all --name whisperlivekit-base -p 8000:8000 whisperlivekit-defaults --model base
-    docker start -i whisperlivekit-base
-    ```

 - `--build-arg` Options:
  - `EXTRAS="whisper-timestamped"` - Add extras to the image's installation (no spaces). Remember to set necessary container options!
@@ -290,10 +264,3 @@ A basic Dockerfile is provided which allows re-use of Python package installatio

 ## 🔮 Use Cases
 Capture discussions in real-time for meeting transcription, help hearing-impaired users follow conversations through accessibility tools, transcribe podcasts or videos automatically for content creation, transcribe support calls with speaker identification for customer service...
-
-## 🙏 Acknowledgments
-
-We extend our gratitude to the original authors of:
-
-| [Whisper Streaming](https://github.com/ufal/whisper_streaming)  | [SimulStreaming](https://github.com/ufal/SimulStreaming) | [Diart](https://github.com/juanmc2005/diart) | [OpenAI Whisper](https://github.com/openai/whisper) |
-| -------- | ------- | -------- | ------- |
--- a/architecture.png
+++ b/architecture.png
--- a/available_models.md
+++ b/available_models.md
@@ -0,0 +1,72 @@
+# Available model sizes:
+
+- tiny.en (english only)
+- tiny
+- base.en (english only)
+- base
+- small.en (english only)
+- small
+- medium.en (english only)
+- medium
+- large-v1
+- large-v2
+- large-v3
+- large-v3-turbo
+
+## How to choose?
+
+### Language Support
+- **English only**: Use `.en` models for better accuracy and faster processing when you only need English transcription
+- **Multilingual**: Do not use `.en` models.
+
+### Resource Constraints
+- **Limited GPU/CPU or need for very low latency**: Choose `small` or smaller models
+  - `tiny`: Fastest, lowest resource usage, acceptable quality for simple audio
+  - `base`: Good balance of speed and accuracy for basic use cases
+  - `small`: Better accuracy while still being resource-efficient
+- **Good resources available**: Use `large` models for best accuracy
+  - `large-v2`: Excellent accuracy, good multilingual support
+  - `large-v3`: Best overall accuracy and language support
+
+### Special Cases
+- **No translation needed**: Use `large-v3-turbo`
+  - Same transcription quality as `large-v2` but significantly faster
+  - **Important**: Does not translate correctly, only transcribes
+
+### Model Comparison Table
+
+| Model | Speed | Accuracy | Multilingual | Translation | Best Use Case |
+|-------|--------|----------|--------------|-------------|---------------|
+| tiny(.en) | Fastest | Basic | Yes/No | Yes/No | Real-time, low resources |
+| base(.en) | Fast | Good | Yes/No | Yes/No | Balanced performance |
+| small(.en) | Medium | Better | Yes/No | Yes/No | Quality on limited hardware |
+| medium(.en) | Slow | High | Yes/No | Yes/No | High quality, moderate resources |
+| large-v2 | Slowest | Excellent | Yes | Yes | Best overall quality |
+| large-v3 | Slowest | Excellent | Yes | Yes | Maximum accuracy |
+| large-v3-turbo | Fast | Excellent | Yes | No | Fast, high-quality transcription |
+
+### Additional Considerations
+
+**Model Performance**:
+- Accuracy improves significantly from tiny to large models
+- English-only models are ~10-15% more accurate for English audio
+- Newer versions (v2, v3) have better punctuation and formatting
+
+**Hardware Requirements**:
+- `tiny`: ~1GB VRAM
+- `base`: ~1GB VRAM  
+- `small`: ~2GB VRAM
+- `medium`: ~5GB VRAM
+- `large`: ~10GB VRAM
+
+**Audio Quality Impact**:
+- Clean, clear audio: smaller models may suffice
+- Noisy, accented, or technical audio: larger models recommended
+- Phone/low-quality audio: use at least `small` model
+
+### Quick Decision Tree
+1. English only? → Add `.en` to your choice
+2. Limited resources or need speed? → `small` or smaller
+3. Good hardware and want best quality? → `large-v3`
+4. Need fast, high-quality transcription without translation? → `large-v3-turbo`
+5. Need translation capabilities? → `large-v2` or `large-v3` (avoid turbo)
--- a/demo.png
+++ b/demo.png
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

 [project]
 name = "whisperlivekit"
-version = "0.2.5"
+version = "0.2.7"
 description = "Real-time, Fully Local Whisper's Speech-to-Text and Speaker Diarization"
 readme = "README.md"
 authors = [
@@ -27,24 +27,16 @@ dependencies = [
    "soundfile",
    "faster-whisper",
    "uvicorn",
-    "websockets"
-]
-
-[project.optional-dependencies]
-diarization = ["diart"]
-vac = ["torch"]
-sentence = ["mosestokenizer", "wtpsplit"]
-whisper = ["whisper"]
-whisper-timestamped = ["whisper-timestamped"]
-mlx-whisper = ["mlx-whisper"]
-openai = ["openai"]
-simulstreaming = [
+    "websockets",
    "torch",
    "tqdm",
    "tiktoken",
-    'triton>=2.0.0,<3; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")'
+    'triton>=2.0.0; platform_machine == "x86_64" and (sys_platform == "linux" or sys_platform == "linux2")'
 ]

+[project.optional-dependencies]
+sentence = ["mosestokenizer", "wtpsplit"]
+
 [project.urls]
 Homepage = "https://github.com/QuentinFuxa/WhisperLiveKit"

@@ -55,5 +47,5 @@ whisperlivekit-server = "whisperlivekit.basic_server:main"
 packages = ["whisperlivekit", "whisperlivekit.diarization", "whisperlivekit.simul_whisper", "whisperlivekit.simul_whisper.whisper", "whisperlivekit.simul_whisper.whisper.assets", "whisperlivekit.simul_whisper.whisper.normalizers", "whisperlivekit.web", "whisperlivekit.whisper_streaming_custom"]

 [tool.setuptools.package-data]
-whisperlivekit = ["web/*.html"]
+whisperlivekit = ["web/*.html", "web/*.css", "web/*.js", "web/src/*.svg"]
 "whisperlivekit.simul_whisper.whisper.assets" = ["*.tiktoken", "*.npz"]
--- a/whisperlivekit/init.py
+++ b/whisperlivekit/init.py
@@ -1,12 +1,13 @@
 from .audio_processor import AudioProcessor
 from .core import TranscriptionEngine
 from .parse_args import parse_args
-from .web.web_interface import get_web_interface_html
+from .web.web_interface import get_web_interface_html, get_inline_ui_html

 __all__ = [
    "TranscriptionEngine",
    "AudioProcessor",
    "parse_args",
    "get_web_interface_html",
+    "get_inline_ui_html",
    "download_simulstreaming_backend",
 ]
--- a/whisperlivekit/audio_processor.py
+++ b/whisperlivekit/audio_processor.py
@@ -4,11 +4,11 @@ from time import time, sleep
 import math
 import logging
 import traceback
-from datetime import timedelta
-from whisperlivekit.timed_objects import ASRToken
-from whisperlivekit.core import TranscriptionEngine, online_factory
+from whisperlivekit.timed_objects import ASRToken, Silence
+from whisperlivekit.core import TranscriptionEngine, online_factory, online_diarization_factory
 from whisperlivekit.ffmpeg_manager import FFmpegManager, FFmpegState
-from .remove_silences import handle_silences
+from whisperlivekit.silero_vad_iterator import FixedVADIterator
+from whisperlivekit.results_formater import format_output, format_time
 # Set up logging once
 logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s")
 logger = logging.getLogger(__name__)
@@ -16,10 +16,6 @@ logger.setLevel(logging.DEBUG)

 SENTINEL = object() # unique sentinel object for end of stream marker

-def format_time(seconds: float) -> str:
-    """Format seconds as HH:MM:SS."""
-    return str(timedelta(seconds=int(seconds)))
-
 class AudioProcessor:
    """
    Processes audio streams for transcription and diarization.
@@ -45,24 +41,31 @@ class AudioProcessor:
        self.last_ffmpeg_activity = time()
        self.ffmpeg_health_check_interval = 5
        self.ffmpeg_max_idle_time = 10
+        self.debug = False

        # State management
        self.is_stopping = False
+        self.silence = False
+        self.silence_duration = 0.0
        self.tokens = []
        self.buffer_transcription = ""
        self.buffer_diarization = ""
        self.end_buffer = 0
        self.end_attributed_speaker = 0
        self.lock = asyncio.Lock()
-        self.beg_loop = time()
+        self.beg_loop = None #to deal with a potential little lag at the websocket initialization, this is now set in process_audio
        self.sep = " "  # Default separator
        self.last_response_content = ""
        
        # Models and processing
        self.asr = models.asr
        self.tokenizer = models.tokenizer
-        self.diarization = models.diarization
-        
+        self.vac_model = models.vac_model
+        if self.args.vac:
+            self.vac = FixedVADIterator(models.vac_model)
+        else:
+            self.vac = None
+            
        self.ffmpeg_manager = FFmpegManager(
            sample_rate=self.sample_rate,
            channels=self.channels
@@ -89,6 +92,11 @@ class AudioProcessor:
        # Initialize transcription engine if enabled
        if self.args.transcription:
            self.online = online_factory(self.args, models.asr, models.tokenizer)
+            
+        # Initialize diarization engine if enabled
+        if self.args.diarization:
+            self.diarization = online_diarization_factory(self.args, models.diarization_model)
+

    def convert_pcm_to_float(self, pcm_buffer):
        """Convert PCM buffer in s16le format to normalized NumPy array."""
@@ -112,7 +120,7 @@ class AudioProcessor:
    async def add_dummy_token(self):
        """Placeholder token when no transcription is available."""
        async with self.lock:
-            current_time = time() - self.beg_loop
+            current_time = time() - self.beg_loop if self.beg_loop else 0
            self.tokens.append(ASRToken(
                start=current_time, end=current_time + 1,
                text=".", speaker=-1, is_dummy=True
@@ -201,18 +209,44 @@ class AudioProcessor:
                    pcm_array = self.convert_pcm_to_float(self.pcm_buffer[:self.max_bytes_per_sec])
                    self.pcm_buffer = self.pcm_buffer[self.max_bytes_per_sec:]
                    
-                    # Send to transcription if enabled
-                    if self.args.transcription and self.transcription_queue:
-                        await self.transcription_queue.put(pcm_array.copy())
+                    res = None
+                    end_of_audio = False
+                    silence_buffer = None
+                    
+                    if self.args.vac:
+                        res = self.vac(pcm_array)
+                    
+                    if res is not None:
+                        if res.get('end', 0) > res.get('start', 0):
+                            end_of_audio = True
+                        elif self.silence: #end of silence
+                            self.silence = False
+                            silence_buffer = Silence(duration=time() - self.start_silence)
+                            
+                    if silence_buffer:
+                        if self.args.transcription and self.transcription_queue:
+                            await self.transcription_queue.put(silence_buffer)
+                        if self.args.diarization and self.diarization_queue:
+                            await self.diarization_queue.put(silence_buffer)

-                    # Send to diarization if enabled
-                    if self.args.diarization and self.diarization_queue:
-                        await self.diarization_queue.put(pcm_array.copy())
+                    if not self.silence:                            
+                        if self.args.transcription and self.transcription_queue:
+                            await self.transcription_queue.put(pcm_array.copy())
+
+                        if self.args.diarization and self.diarization_queue:
+                            await self.diarization_queue.put(pcm_array.copy())
+                        
+                        self.silence_duration = 0.0
+                        if end_of_audio:
+                            self.silence = True
+                            self.start_silence = time()

                    # Sleep if no processing is happening
                    if not self.args.transcription and not self.args.diarization:
                        await asyncio.sleep(0.1)
                    
+                    
+                    
            except Exception as e:
                logger.warning(f"Exception in ffmpeg_stdout_reader: {e}")
                logger.warning(f"Traceback: {traceback.format_exc()}")
@@ -239,8 +273,8 @@ class AudioProcessor:
        
        while True:
            try:
-                pcm_array = await self.transcription_queue.get()
-                if pcm_array is SENTINEL:
+                item = await self.transcription_queue.get()
+                if item is SENTINEL:
                    logger.debug("Transcription processor received sentinel. Finishing.")
                    self.transcription_queue.task_done()
                    break
@@ -252,17 +286,30 @@ class AudioProcessor:

                asr_internal_buffer_duration_s = len(getattr(self.online, 'audio_buffer', [])) / self.online.SAMPLING_RATE
                transcription_lag_s = max(0.0, time() - self.beg_loop - self.end_buffer)
-
-                logger.info(
-                    f"ASR processing: internal_buffer={asr_internal_buffer_duration_s:.2f}s, "
-                    f"lag={transcription_lag_s:.2f}s."
-                )
+                asr_processing_logs = f"internal_buffer={asr_internal_buffer_duration_s:.2f}s | lag={transcription_lag_s:.2f}s |"
+                if type(item) is Silence:
+                    asr_processing_logs += f" + Silence of = {item.duration:.2f}s"
+                    if self.tokens:
+                        asr_processing_logs += f" | last_end = {self.tokens[-1].end} |"
+                logger.info(asr_processing_logs)
                
-                # Process transcription
-                duration_this_chunk = len(pcm_array) / self.sample_rate if isinstance(pcm_array, np.ndarray) else 0
+                if type(item) is Silence:
+                    cumulative_pcm_duration_stream_time += item.duration
+                    self.online.insert_silence(item.duration, self.tokens[-1].end if self.tokens else 0)
+                    continue
+                
+                if isinstance(item, np.ndarray):
+                    pcm_array = item
+                else:
+                    raise Exception('item should be pcm_array')
+                
+                duration_this_chunk = len(pcm_array) / self.sample_rate
                cumulative_pcm_duration_stream_time += duration_this_chunk
                stream_time_end_of_current_pcm = cumulative_pcm_duration_stream_time

+                
+                    
+
                self.online.insert_audio_chunk(pcm_array, stream_time_end_of_current_pcm)
                new_tokens, current_audio_processed_upto = self.online.process_iter()
                
@@ -303,15 +350,25 @@ class AudioProcessor:
    async def diarization_processor(self, diarization_obj):
        """Process audio chunks for speaker diarization."""
        buffer_diarization = ""
-        
+        cumulative_pcm_duration_stream_time = 0.0
        while True:
            try:
-                pcm_array = await self.diarization_queue.get()
-                if pcm_array is SENTINEL:
+                item = await self.diarization_queue.get()
+                if item is SENTINEL:
                    logger.debug("Diarization processor received sentinel. Finishing.")
                    self.diarization_queue.task_done()
                    break
                
+                if type(item) is Silence:
+                    cumulative_pcm_duration_stream_time += item.duration
+                    diarization_obj.insert_silence(item.duration)
+                    continue
+    
+                if isinstance(item, np.ndarray):
+                    pcm_array = item
+                else:
+                    raise Exception('item should be pcm_array') 
+                
                # Process diarization
                await diarization_obj.diarize(pcm_array)
                
@@ -363,7 +420,7 @@ class AudioProcessor:
                buffer_diarization = state["buffer_diarization"]
                end_attributed_speaker = state["end_attributed_speaker"]
                sep = state["sep"]
-                
+                                
                # Add dummy tokens if needed
                if (not tokens or tokens[-1].is_dummy) and not self.args.transcription and self.args.diarization:
                    await self.add_dummy_token()
@@ -372,40 +429,13 @@ class AudioProcessor:
                    tokens = state["tokens"]
                
                # Format output
-                previous_speaker = -1
-                lines = []
-                last_end_diarized = 0
-                undiarized_text = []
-                current_time = time() - self.beg_loop
-                tokens = handle_silences(tokens, current_time)
-                for token in tokens:
-                    speaker = token.speaker
-                    
-                    # Handle diarization
-                    if self.args.diarization:
-                        if (speaker in [-1, 0]) and token.end >= end_attributed_speaker:
-                            undiarized_text.append(token.text)
-                            continue
-                        elif (speaker in [-1, 0]) and token.end < end_attributed_speaker:
-                            speaker = previous_speaker
-                        if speaker not in [-1, 0]:
-                            last_end_diarized = max(token.end, last_end_diarized)
-
-                    # Group by speaker
-                    if speaker != previous_speaker or not lines:
-                        lines.append({
-                            "speaker": speaker,
-                            "text": token.text,
-                            "beg": format_time(token.start),
-                            "end": format_time(token.end),
-                            "diff": round(token.end - last_end_diarized, 2)
-                        })
-                        previous_speaker = speaker
-                    elif token.text:  # Only append if text isn't empty
-                        lines[-1]["text"] += sep + token.text
-                        lines[-1]["end"] = format_time(token.end)
-                        lines[-1]["diff"] = round(token.end - last_end_diarized, 2)
-                
+                lines, undiarized_text, buffer_transcription, buffer_diarization = format_output(
+                    state,
+                    self.silence,
+                    current_time = time() - self.beg_loop if self.beg_loop else None,
+                    diarization = self.args.diarization,
+                    debug = self.debug
+                )
                # Handle undiarized text
                if undiarized_text:
                    combined = sep.join(undiarized_text)
@@ -435,7 +465,7 @@ class AudioProcessor:
                    "buffer_transcription": buffer_transcription,
                    "buffer_diarization": buffer_diarization,
                    "remaining_time_transcription": state["remaining_time_transcription"],
-                    "remaining_time_diarization": state["remaining_time_diarization"]
+                    "remaining_time_diarization": state["remaining_time_diarization"] if self.args.diarization else 0
                }
                
                current_response_signature = f"{response_status} | " + \
@@ -566,6 +596,10 @@ class AudioProcessor:

    async def process_audio(self, message):
        """Process incoming audio data."""
+
+        if not self.beg_loop:
+            self.beg_loop = time()
+
        if not message:
            logger.info("Empty audio message received, initiating stop sequence.")
            self.is_stopping = True
--- a/whisperlivekit/basic_server.py
+++ b/whisperlivekit/basic_server.py
@@ -2,9 +2,12 @@ from contextlib import asynccontextmanager
 from fastapi import FastAPI, WebSocket, WebSocketDisconnect
 from fastapi.responses import HTMLResponse
 from fastapi.middleware.cors import CORSMiddleware
-from whisperlivekit import TranscriptionEngine, AudioProcessor, get_web_interface_html, parse_args
+from whisperlivekit import TranscriptionEngine, AudioProcessor, get_inline_ui_html, parse_args
 import asyncio
 import logging
+from starlette.staticfiles import StaticFiles
+import pathlib
+import whisperlivekit.web as webpkg

 logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s")
 logging.getLogger().setLevel(logging.WARNING)
@@ -30,10 +33,12 @@ app.add_middleware(
    allow_methods=["*"],
    allow_headers=["*"],
 )
+web_dir = pathlib.Path(webpkg.__file__).parent
+app.mount("/web", StaticFiles(directory=str(web_dir)), name="web")

@app.get("/")
 async def get():
-    return HTMLResponse(get_web_interface_html())
+    return HTMLResponse(get_inline_ui_html())


 async def handle_websocket_results(websocket, results_generator):
@@ -47,7 +52,7 @@ async def handle_websocket_results(websocket, results_generator):
    except WebSocketDisconnect:
        logger.info("WebSocket disconnected while handling results (client likely closed connection).")
    except Exception as e:
-        logger.warning(f"Error in WebSocket results handler: {e}")
+        logger.exception(f"Error in WebSocket results handler: {e}")


@app.websocket("/asr")
--- a/whisperlivekit/core.py
+++ b/whisperlivekit/core.py
@@ -1,9 +1,9 @@
 try:
    from whisperlivekit.whisper_streaming_custom.whisper_online import backend_factory
-    from whisperlivekit.whisper_streaming_custom.online_asr import VACOnlineASRProcessor, OnlineASRProcessor
+    from whisperlivekit.whisper_streaming_custom.online_asr import OnlineASRProcessor
 except ImportError:
    from .whisper_streaming_custom.whisper_online import backend_factory
-    from .whisper_streaming_custom.online_asr import VACOnlineASRProcessor, OnlineASRProcessor
+    from .whisper_streaming_custom.online_asr import OnlineASRProcessor
 from whisperlivekit.warmup import warmup_asr, warmup_online
 from argparse import Namespace
 import sys
@@ -34,7 +34,7 @@ class TranscriptionEngine:
            "lan": "auto",
            "task": "transcribe",
            "backend": "faster-whisper",
-            "vac": False,
+            "vac": True,
            "vac_chunk_size": 0.04,
            "log_level": "DEBUG",
            "ssl_certfile": None,
@@ -49,7 +49,7 @@ class TranscriptionEngine:
            "frame_threshold": 25,
            "beams": 1,
            "decoder_type": None,
-            "audio_max_len": 30.0,
+            "audio_max_len": 20.0,
            "audio_min_len": 0.0,
            "cif_ckpt_path": None,
            "never_fire": False,
@@ -57,10 +57,10 @@ class TranscriptionEngine:
            "static_init_prompt": None,
            "max_context_tokens": None,
            "model_path": './base.pt',
+            "diarization_backend": "sortformer",
            # diart params:
            "segmentation_model": "pyannote/segmentation-3.0",
            "embedding_model": "pyannote/embedding",
-
        }

        config_dict = {**defaults, **kwargs}
@@ -69,6 +69,8 @@ class TranscriptionEngine:
            config_dict['transcription'] = not kwargs['no_transcription']
        if 'no_vad' in kwargs:
            config_dict['vad'] = not kwargs['no_vad']
+        if 'no_vac' in kwargs:
+            config_dict['vac'] = not kwargs['no_vac']
        
        config_dict.pop('no_transcription', None)
        config_dict.pop('no_vad', None)
@@ -82,15 +84,20 @@ class TranscriptionEngine:
        self.asr = None
        self.tokenizer = None
        self.diarization = None
+        self.vac_model = None
+        
+        if self.args.vac:
+            import torch
+            self.vac_model, _ = torch.hub.load(repo_or_dir="snakers4/silero-vad", model="silero_vad")            
        
        if self.args.transcription:
            if self.args.backend == "simulstreaming": 
-                from simul_whisper import SimulStreamingASR
+                from whisperlivekit.simul_whisper import SimulStreamingASR
                self.tokenizer = None
                simulstreaming_kwargs = {}
                for attr in ['frame_threshold', 'beams', 'decoder_type', 'audio_max_len', 'audio_min_len', 
                            'cif_ckpt_path', 'never_fire', 'init_prompt', 'static_init_prompt', 
-                            'max_context_tokens', 'model_path']:
+                            'max_context_tokens', 'model_path', 'warmup_file', 'preload_model_count']:
                    if hasattr(self.args, attr):
                        simulstreaming_kwargs[attr] = getattr(self.args, attr)
        
@@ -112,12 +119,18 @@ class TranscriptionEngine:
            warmup_asr(self.asr, self.args.warmup_file) #for simulstreaming, warmup should be done in the online class not here

        if self.args.diarization:
-            from whisperlivekit.diarization.diarization_online import DiartDiarization
-            self.diarization = DiartDiarization(
-                block_duration=self.args.min_chunk_size,
-                segmentation_model_name=self.args.segmentation_model,
-                embedding_model_name=self.args.embedding_model
-            )
+            if self.args.diarization_backend == "diart":
+                from whisperlivekit.diarization.diart_backend import DiartDiarization
+                self.diarization_model = DiartDiarization(
+                    block_duration=self.args.min_chunk_size,
+                    segmentation_model_name=self.args.segmentation_model,
+                    embedding_model_name=self.args.embedding_model
+                )
+            elif self.args.diarization_backend == "sortformer":
+                from whisperlivekit.diarization.sortformer_backend import SortformerDiarization
+                self.diarization_model = SortformerDiarization()
+            else:
+                raise ValueError(f"Unknown diarization backend: {self.args.diarization_backend}")
            
        TranscriptionEngine._initialized = True

@@ -125,21 +138,12 @@ class TranscriptionEngine:

 def online_factory(args, asr, tokenizer, logfile=sys.stderr):
    if args.backend == "simulstreaming":    
-        from simul_whisper import SimulStreamingOnlineProcessor
+        from whisperlivekit.simul_whisper import SimulStreamingOnlineProcessor
        online = SimulStreamingOnlineProcessor(
            asr,
            logfile=logfile,
        )
        # warmup_online(online, args.warmup_file)
-    elif args.vac:
-        online = VACOnlineASRProcessor(
-            args.min_chunk_size,
-            asr,
-            tokenizer,
-            logfile=logfile,
-            buffer_trimming=(args.buffer_trimming, args.buffer_trimming_sec),
-            confidence_validation = args.confidence_validation
-        )
    else:
        online = OnlineASRProcessor(
            asr,
@@ -149,4 +153,16 @@ def online_factory(args, asr, tokenizer, logfile=sys.stderr):
            confidence_validation = args.confidence_validation
        )
    return online
-  
+  
+  
+def online_diarization_factory(args, diarization_backend):
+    if args.diarization_backend == "diart":
+        online = diarization_backend
+        # Not the best here, since several user/instances will share the same backend, but diart is not SOTA anymore and sortformer is recommanded
+    
+    if args.diarization_backend == "sortformer":
+        from whisperlivekit.diarization.sortformer_backend import SortformerDiarizationOnline
+        online = SortformerDiarizationOnline(shared_model=diarization_backend)
+    return online
+
+        
--- a/whisperlivekit/diarization/diarization_online.py
+++ b/whisperlivekit/diarization/diarization_online.py
@@ -29,6 +29,7 @@ class DiarizationObserver(Observer):
        self.speaker_segments = []
        self.processed_time = 0
        self.segment_lock = threading.Lock()
+        self.global_time_offset = 0.0
    
    def on_next(self, value: Tuple[Annotation, Any]):
        annotation, audio = value
@@ -49,8 +50,8 @@ class DiarizationObserver(Observer):
                        print(f"  {speaker}: {start:.2f}s-{end:.2f}s")
                        self.speaker_segments.append(SpeakerSegment(
                            speaker=speaker,
-                            start=start,
-                            end=end
+                            start=start + self.global_time_offset,
+                            end=end + self.global_time_offset
                        ))
            else:
                logger.debug("\nNo speakers detected in this segment")
@@ -199,6 +200,9 @@ class DiartDiarization:
        self.inference.attach_observers(self.observer)
        asyncio.get_event_loop().run_in_executor(None, self.inference)

+    def insert_silence(self, silence_duration):
+        self.observer.global_time_offset += silence_duration
+
    async def diarize(self, pcm_array: np.ndarray):
        """
        Process audio data for diarization.
--- a/whisperlivekit/diarization/sortformer_backend.py
+++ b/whisperlivekit/diarization/sortformer_backend.py
@@ -0,0 +1,457 @@
+import numpy as np
+import torch
+import logging
+import threading
+import time
+import wave
+from typing import List, Optional
+from queue import SimpleQueue, Empty
+
+from whisperlivekit.timed_objects import SpeakerSegment
+
+logger = logging.getLogger(__name__)
+
+try:
+    from nemo.collections.asr.models import SortformerEncLabelModel
+    from nemo.collections.asr.modules import AudioToMelSpectrogramPreprocessor
+except ImportError:
+    raise SystemExit("""Please use `pip install "git+https://github.com/NVIDIA/NeMo.git@main#egg=nemo_toolkit[asr]"` to use the Sortformer diarization""")
+
+
+class StreamingSortformerState:
+    """
+    This class creates a class instance that will be used to store the state of the
+    streaming Sortformer model.
+
+    Attributes:
+        spkcache (torch.Tensor): Speaker cache to store embeddings from start
+        spkcache_lengths (torch.Tensor): Lengths of the speaker cache
+        spkcache_preds (torch.Tensor): The speaker predictions for the speaker cache parts
+        fifo (torch.Tensor): FIFO queue to save the embedding from the latest chunks
+        fifo_lengths (torch.Tensor): Lengths of the FIFO queue
+        fifo_preds (torch.Tensor): The speaker predictions for the FIFO queue parts
+        spk_perm (torch.Tensor): Speaker permutation information for the speaker cache
+        mean_sil_emb (torch.Tensor): Mean silence embedding
+        n_sil_frames (torch.Tensor): Number of silence frames
+    """
+
+    def __init__(self):
+        self.spkcache = None  # Speaker cache to store embeddings from start
+        self.spkcache_lengths = None
+        self.spkcache_preds = None  # speaker cache predictions
+        self.fifo = None  # to save the embedding from the latest chunks
+        self.fifo_lengths = None
+        self.fifo_preds = None
+        self.spk_perm = None
+        self.mean_sil_emb = None
+        self.n_sil_frames = None
+
+
+class SortformerDiarization:
+    def __init__(self, model_name: str = "nvidia/diar_streaming_sortformer_4spk-v2"):
+        """
+        Stores the shared streaming Sortformer diarization model. Used when a new online_diarization is initialized.
+        """
+        self._load_model(model_name)
+    
+    def _load_model(self, model_name: str):
+        """Load and configure the Sortformer model for streaming."""
+        try:
+            self.diar_model = SortformerEncLabelModel.from_pretrained(model_name)
+            self.diar_model.eval()
+
+            if torch.cuda.is_available():
+                self.diar_model.to(torch.device("cuda"))
+                logger.info("Using CUDA for Sortformer model")
+            else:
+                logger.info("Using CPU for Sortformer model")
+
+            self.diar_model.sortformer_modules.chunk_len = 10
+            self.diar_model.sortformer_modules.subsampling_factor = 10
+            self.diar_model.sortformer_modules.chunk_right_context = 0
+            self.diar_model.sortformer_modules.chunk_left_context = 10
+            self.diar_model.sortformer_modules.spkcache_len = 188
+            self.diar_model.sortformer_modules.fifo_len = 188
+            self.diar_model.sortformer_modules.spkcache_update_period = 144
+            self.diar_model.sortformer_modules.log = False
+            self.diar_model.sortformer_modules._check_streaming_parameters()
+                        
+        except Exception as e:
+            logger.error(f"Failed to load Sortformer model: {e}")
+            raise
+ 
+class SortformerDiarizationOnline:
+    def __init__(self, shared_model, sample_rate: int = 16000):
+        """
+        Initialize the streaming Sortformer diarization system.
+        
+        Args:
+            sample_rate: Audio sample rate (default: 16000)
+            model_name: Pre-trained model name (default: "nvidia/diar_streaming_sortformer_4spk-v2")
+        """
+        self.sample_rate = sample_rate
+        self.speaker_segments = []
+        self.buffer_audio = np.array([], dtype=np.float32)
+        self.segment_lock = threading.Lock()
+        self.global_time_offset = 0.0
+        self.processed_time = 0.0
+        self.debug = False
+                
+        self.diar_model = shared_model.diar_model
+             
+        self.audio2mel = AudioToMelSpectrogramPreprocessor(
+            window_size=0.025,
+            normalize="NA",
+            n_fft=512,
+            features=128,
+            pad_to=0
+        )
+        
+        self.chunk_duration_seconds = (
+            self.diar_model.sortformer_modules.chunk_len * 
+            self.diar_model.sortformer_modules.subsampling_factor * 
+            self.diar_model.preprocessor._cfg.window_stride
+        )
+        
+        self._init_streaming_state()
+        
+        self._previous_chunk_features = None
+        self._chunk_index = 0
+        self._len_prediction = None
+        
+        # Audio buffer to store PCM chunks for debugging
+        self.audio_buffer = []
+        
+        # Buffer for accumulating audio chunks until reaching chunk_duration_seconds
+        self.audio_chunk_buffer = []
+        self.accumulated_duration = 0.0
+        
+        logger.info("SortformerDiarization initialized successfully")
+
+
+    def _init_streaming_state(self):
+        """Initialize the streaming state for the model."""
+        batch_size = 1
+        device = self.diar_model.device
+        
+        self.streaming_state = StreamingSortformerState()
+        self.streaming_state.spkcache = torch.zeros(
+            (batch_size, self.diar_model.sortformer_modules.spkcache_len, self.diar_model.sortformer_modules.fc_d_model), 
+            device=device
+        )
+        self.streaming_state.spkcache_preds = torch.zeros(
+            (batch_size, self.diar_model.sortformer_modules.spkcache_len, self.diar_model.sortformer_modules.n_spk), 
+            device=device
+        )
+        self.streaming_state.spkcache_lengths = torch.zeros((batch_size,), dtype=torch.long, device=device)
+        self.streaming_state.fifo = torch.zeros(
+            (batch_size, self.diar_model.sortformer_modules.fifo_len, self.diar_model.sortformer_modules.fc_d_model), 
+            device=device
+        )
+        self.streaming_state.fifo_lengths = torch.zeros((batch_size,), dtype=torch.long, device=device)
+        self.streaming_state.mean_sil_emb = torch.zeros((batch_size, self.diar_model.sortformer_modules.fc_d_model), device=device)
+        self.streaming_state.n_sil_frames = torch.zeros((batch_size,), dtype=torch.long, device=device)
+        
+        # Initialize total predictions tensor
+        self.total_preds = torch.zeros((batch_size, 0, self.diar_model.sortformer_modules.n_spk), device=device)
+
+    def insert_silence(self, silence_duration: float):
+        """
+        Insert silence period by adjusting the global time offset.
+        
+        Args:
+            silence_duration: Duration of silence in seconds
+        """
+        with self.segment_lock:
+            self.global_time_offset += silence_duration
+        logger.debug(f"Inserted silence of {silence_duration:.2f}s, new offset: {self.global_time_offset:.2f}s")
+
+    async def diarize(self, pcm_array: np.ndarray):
+        """
+        Process audio data for diarization in streaming fashion.
+        
+        Args:
+            pcm_array: Audio data as numpy array
+        """
+        try:
+            if self.debug:
+                self.audio_buffer.append(pcm_array.copy())
+
+            threshold = int(self.chunk_duration_seconds * self.sample_rate)
+            
+            self.buffer_audio = np.concatenate([self.buffer_audio, pcm_array.copy()])
+            if not len(self.buffer_audio) >= threshold:
+                return
+            
+            audio = self.buffer_audio[:threshold]
+            self.buffer_audio = self.buffer_audio[threshold:]
+            
+            audio_signal_chunk = torch.tensor(audio).unsqueeze(0).to(self.diar_model.device)
+            audio_signal_length_chunk = torch.tensor([audio_signal_chunk.shape[1]]).to(self.diar_model.device)
+            
+            processed_signal_chunk, processed_signal_length_chunk = self.audio2mel.get_features(
+                audio_signal_chunk, audio_signal_length_chunk
+            )
+            
+            if self._previous_chunk_features is not None:
+                to_add = self._previous_chunk_features[:, :, -99:]
+                total_features = torch.concat([to_add, processed_signal_chunk], dim=2)
+            else:
+                total_features = processed_signal_chunk
+            
+            self._previous_chunk_features = processed_signal_chunk
+            
+            chunk_feat_seq_t = torch.transpose(total_features, 1, 2)
+            
+            with torch.inference_mode():
+                left_offset = 8 if self._chunk_index > 0 else 0
+                right_offset = 8
+                
+                self.streaming_state, self.total_preds = self.diar_model.forward_streaming_step(
+                    processed_signal=chunk_feat_seq_t,
+                    processed_signal_length=torch.tensor([chunk_feat_seq_t.shape[1]]),
+                    streaming_state=self.streaming_state,
+                    total_preds=self.total_preds,
+                    left_offset=left_offset,
+                    right_offset=right_offset,
+                )
+                
+            # Convert predictions to speaker segments
+            self._process_predictions()
+            
+            self._chunk_index += 1
+            
+        except Exception as e:
+            logger.error(f"Error in diarize: {e}")
+            raise
+            
+        # TODO: Handle case when stream ends with partial buffer (accumulated_duration > 0 but < chunk_duration_seconds)
+
+    def _process_predictions(self):
+        """Process model predictions and convert to speaker segments."""
+        try:
+            preds_np = self.total_preds[0].cpu().numpy()
+            active_speakers = np.argmax(preds_np, axis=1)
+            
+            if self._len_prediction is None:
+                self._len_prediction = len(active_speakers)
+            
+            # Get predictions for current chunk
+            frame_duration = self.chunk_duration_seconds / self._len_prediction
+            current_chunk_preds = active_speakers[-self._len_prediction:]
+            
+            with self.segment_lock:
+                # Process predictions into segments
+                base_time = self._chunk_index * self.chunk_duration_seconds + self.global_time_offset
+                
+                for idx, spk in enumerate(current_chunk_preds):
+                    start_time = base_time + idx * frame_duration
+                    end_time = base_time + (idx + 1) * frame_duration
+                    
+                    # Check if this continues the last segment or starts a new one
+                    if (self.speaker_segments and 
+                        self.speaker_segments[-1].speaker == spk and 
+                        abs(self.speaker_segments[-1].end - start_time) < frame_duration * 0.5):
+                        # Continue existing segment
+                        self.speaker_segments[-1].end = end_time
+                    else:
+                        
+                        # Create new segment
+                        self.speaker_segments.append(SpeakerSegment(
+                            speaker=spk,
+                            start=start_time,
+                            end=end_time
+                        ))
+                
+                # Update processed time
+                self.processed_time = max(self.processed_time, base_time + self.chunk_duration_seconds)
+                
+                logger.debug(f"Processed chunk {self._chunk_index}, total segments: {len(self.speaker_segments)}")
+                
+        except Exception as e:
+            logger.error(f"Error processing predictions: {e}")
+
+    def assign_speakers_to_tokens(self, tokens: list, use_punctuation_split: bool = False) -> list:
+        """
+        Assign speakers to tokens based on timing overlap with speaker segments.
+        
+        Args:
+            tokens: List of tokens with timing information
+            use_punctuation_split: Whether to use punctuation for boundary refinement
+            
+        Returns:
+            List of tokens with speaker assignments
+        """
+        with self.segment_lock:
+            segments = self.speaker_segments.copy()
+        
+        if not segments or not tokens:
+            logger.debug("No segments or tokens available for speaker assignment")
+            return tokens
+        
+        logger.debug(f"Assigning speakers to {len(tokens)} tokens using {len(segments)} segments")
+        use_punctuation_split = False
+        if not use_punctuation_split:
+            # Simple overlap-based assignment
+            for token in tokens:
+                token.speaker = -1  # Default to no speaker
+                for segment in segments:
+                    # Check for timing overlap
+                    if not (segment.end <= token.start or segment.start >= token.end):
+                        token.speaker = segment.speaker + 1  # Convert to 1-based indexing
+                        break
+        else:
+            # Use punctuation-aware assignment (similar to diart_backend)
+            tokens = self._add_speaker_to_tokens_with_punctuation(segments, tokens)
+        
+        return tokens
+
+    def _add_speaker_to_tokens_with_punctuation(self, segments: List[SpeakerSegment], tokens: list) -> list:
+        """
+        Assign speakers to tokens with punctuation-aware boundary adjustment.
+        
+        Args:
+            segments: List of speaker segments
+            tokens: List of tokens to assign speakers to
+            
+        Returns:
+            List of tokens with speaker assignments
+        """
+        punctuation_marks = {'.', '!', '?'}
+        punctuation_tokens = [token for token in tokens if token.text.strip() in punctuation_marks]
+        
+        # Convert segments to concatenated format
+        segments_concatenated = self._concatenate_speakers(segments)
+        
+        # Adjust segment boundaries based on punctuation
+        for ind, segment in enumerate(segments_concatenated):
+            for i, punctuation_token in enumerate(punctuation_tokens):
+                if punctuation_token.start > segment['end']:
+                    after_length = punctuation_token.start - segment['end']
+                    before_length = segment['end'] - punctuation_tokens[i - 1].end if i > 0 else float('inf')
+                    
+                    if before_length > after_length:
+                        segment['end'] = punctuation_token.start
+                        if i < len(punctuation_tokens) - 1 and ind + 1 < len(segments_concatenated):
+                            segments_concatenated[ind + 1]['begin'] = punctuation_token.start
+                    else:
+                        segment['end'] = punctuation_tokens[i - 1].end if i > 0 else segment['end']
+                        if i < len(punctuation_tokens) - 1 and ind - 1 >= 0:
+                            segments_concatenated[ind - 1]['begin'] = punctuation_tokens[i - 1].end
+                    break
+        
+        # Ensure non-overlapping tokens
+        last_end = 0.0
+        for token in tokens:
+            start = max(last_end + 0.01, token.start)
+            token.start = start
+            token.end = max(start, token.end)
+            last_end = token.end
+        
+        # Assign speakers based on adjusted segments
+        ind_last_speaker = 0
+        for segment in segments_concatenated:
+            for i, token in enumerate(tokens[ind_last_speaker:]):
+                if token.end <= segment['end']:
+                    token.speaker = segment['speaker']
+                    ind_last_speaker = i + 1
+                elif token.start > segment['end']:
+                    break
+        
+        return tokens
+
+    def _concatenate_speakers(self, segments: List[SpeakerSegment]) -> List[dict]:
+        """
+        Concatenate consecutive segments from the same speaker.
+        
+        Args:
+            segments: List of speaker segments
+            
+        Returns:
+            List of concatenated speaker segments
+        """
+        if not segments:
+            return []
+            
+        segments_concatenated = [{"speaker": segments[0].speaker + 1, "begin": segments[0].start, "end": segments[0].end}]
+        
+        for segment in segments[1:]:
+            speaker = segment.speaker + 1
+            if segments_concatenated[-1]['speaker'] != speaker:
+                segments_concatenated.append({"speaker": speaker, "begin": segment.start, "end": segment.end})
+            else:
+                segments_concatenated[-1]['end'] = segment.end
+                
+        return segments_concatenated
+
+    def get_segments(self) -> List[SpeakerSegment]:
+        """Get a copy of the current speaker segments."""
+        with self.segment_lock:
+            return self.speaker_segments.copy()
+
+    def clear_old_segments(self, older_than: float = 30.0):
+        """Clear segments older than the specified time."""
+        with self.segment_lock:
+            current_time = self.processed_time
+            self.speaker_segments = [
+                segment for segment in self.speaker_segments 
+                if current_time - segment.end < older_than
+            ]
+            logger.debug(f"Cleared old segments, remaining: {len(self.speaker_segments)}")
+
+    def close(self):
+        """Close the diarization system and clean up resources."""
+        logger.info("Closing SortformerDiarization")
+        with self.segment_lock:
+            self.speaker_segments.clear()
+        
+        if self.debug:
+            concatenated_audio = np.concatenate(self.audio_buffer)
+            audio_data_int16 = (concatenated_audio * 32767).astype(np.int16)                
+            with wave.open("diarization_audio.wav", "wb") as wav_file:
+                wav_file.setnchannels(1)  # mono audio
+                wav_file.setsampwidth(2)   # 2 bytes per sample (int16)
+                wav_file.setframerate(self.sample_rate)
+                wav_file.writeframes(audio_data_int16.tobytes())
+            logger.info(f"Saved {len(concatenated_audio)} samples to diarization_audio.wav")
+
+
+def extract_number(s: str) -> int:
+    """Extract number from speaker string (compatibility function)."""
+    import re
+    m = re.search(r'\d+', s)
+    return int(m.group()) if m else 0
+
+
+if __name__ == '__main__':
+    import asyncio
+    import librosa
+    
+    async def main():
+        """TEST ONLY."""
+        an4_audio = 'audio_test.mp3'
+        signal, sr = librosa.load(an4_audio, sr=16000)
+        signal = signal[:16000*30]
+
+        print("\n" + "=" * 50)
+        print("ground truth:")
+        print("Speaker 0: 0:00 - 0:09")
+        print("Speaker 1: 0:09 - 0:19") 
+        print("Speaker 2: 0:19 - 0:25")
+        print("Speaker 0: 0:25 - 0:30")
+        print("=" * 50)
+        
+        diarization = SortformerDiarization(sample_rate=16000)        
+        chunk_size = 1600
+        
+        for i in range(0, len(signal), chunk_size):
+            chunk = signal[i:i+chunk_size]
+            await diarization.diarize(chunk)
+            print(f"Processed chunk {i // chunk_size + 1}")
+        
+        segments = diarization.get_segments()
+        print("\nDiarization results:")
+        for segment in segments:
+            print(f"Speaker {segment.speaker}: {segment.start:.2f}s - {segment.end:.2f}s")
+    
+    asyncio.run(main())
--- a/whisperlivekit/diarization/sortformer_backend_offline.py
+++ b/whisperlivekit/diarization/sortformer_backend_offline.py
@@ -0,0 +1,205 @@
+import numpy as np
+import torch
+import logging
+
+from nemo.collections.asr.models import SortformerEncLabelModel
+from nemo.collections.asr.modules import AudioToMelSpectrogramPreprocessor
+import librosa
+
+logger = logging.getLogger(__name__)
+
+def load_model():
+
+    diar_model = SortformerEncLabelModel.from_pretrained("nvidia/diar_streaming_sortformer_4spk-v2")
+    diar_model.eval()
+
+    if torch.cuda.is_available():
+        diar_model.to(torch.device("cuda"))
+
+    #we target 1 second lag for the moment. chunk_len could be reduced.
+    diar_model.sortformer_modules.chunk_len = 10
+    diar_model.sortformer_modules.subsampling_factor = 10 #8 would be better ideally
+
+    diar_model.sortformer_modules.chunk_right_context = 0 #no.
+    diar_model.sortformer_modules.chunk_left_context = 10 #big so it compensiate the problem with no padding later.
+
+    diar_model.sortformer_modules.spkcache_len = 188
+    diar_model.sortformer_modules.fifo_len = 188
+    diar_model.sortformer_modules.spkcache_update_period = 144
+    diar_model.sortformer_modules.log = False
+    diar_model.sortformer_modules._check_streaming_parameters()
+
+
+    audio2mel = AudioToMelSpectrogramPreprocessor(
+            window_size= 0.025, 
+            normalize="NA",
+            n_fft=512,
+            features=128,
+            pad_to=0) #pad_to 16 works better than 0. On test audio, we detect a third speaker for 1 second with pad_to=0. To solve that : increase left context to 10.
+
+    return diar_model, audio2mel
+
+diar_model, audio2mel = load_model()
+
+class StreamingSortformerState:
+    """
+    This class creates a class instance that will be used to store the state of the
+    streaming Sortformer model.
+
+    Attributes:
+        spkcache (torch.Tensor): Speaker cache to store embeddings from start
+        spkcache_lengths (torch.Tensor): Lengths of the speaker cache
+        spkcache_preds (torch.Tensor): The speaker predictions for the speaker cache parts
+        fifo (torch.Tensor): FIFO queue to save the embedding from the latest chunks
+        fifo_lengths (torch.Tensor): Lengths of the FIFO queue
+        fifo_preds (torch.Tensor): The speaker predictions for the FIFO queue parts
+        spk_perm (torch.Tensor): Speaker permutation information for the speaker cache
+        mean_sil_emb (torch.Tensor): Mean silence embedding
+        n_sil_frames (torch.Tensor): Number of silence frames
+    """
+
+    spkcache = None  # Speaker cache to store embeddings from start
+    spkcache_lengths = None  #
+    spkcache_preds = None  # speaker cache predictions
+    fifo = None  # to save the embedding from the latest chunks
+    fifo_lengths = None
+    fifo_preds = None
+    spk_perm = None
+    mean_sil_emb = None
+    n_sil_frames = None
+
+
+def init_streaming_state(self, batch_size: int = 1, async_streaming: bool = False, device: torch.device = None):
+    """
+    Initializes StreamingSortformerState with empty tensors or zero-valued tensors.
+
+    Args:
+        batch_size (int): Batch size for tensors in streaming state
+        async_streaming (bool): True for asynchronous update, False for synchronous update
+        device (torch.device): Device for tensors in streaming state
+
+    Returns:
+        streaming_state (SortformerStreamingState): initialized streaming state
+    """
+    streaming_state = StreamingSortformerState()
+    if async_streaming:
+        streaming_state.spkcache = torch.zeros((batch_size, self.spkcache_len, self.fc_d_model), device=device)
+        streaming_state.spkcache_preds = torch.zeros((batch_size, self.spkcache_len, self.n_spk), device=device)
+        streaming_state.spkcache_lengths = torch.zeros((batch_size,), dtype=torch.long, device=device)
+        streaming_state.fifo = torch.zeros((batch_size, self.fifo_len, self.fc_d_model), device=device)
+        streaming_state.fifo_lengths = torch.zeros((batch_size,), dtype=torch.long, device=device)
+    else:
+        streaming_state.spkcache = torch.zeros((batch_size, 0, self.fc_d_model), device=device)
+        streaming_state.fifo = torch.zeros((batch_size, 0, self.fc_d_model), device=device)
+    streaming_state.mean_sil_emb = torch.zeros((batch_size, self.fc_d_model), device=device)
+    streaming_state.n_sil_frames = torch.zeros((batch_size,), dtype=torch.long, device=device)
+    return streaming_state
+
+
+def process_diarization(chunks):
+    """ 
+    what it does:
+    1. Preprocessing: Applies dithering and pre-emphasis (high-pass filter) if enabled
+    2. STFT: Computes the Short-Time Fourier Transform using:
+        - the window of window_size=0.025 --> size of a window : 400 samples
+        - the hop parameter : n_window_stride = 0.01 -> every 160 samples, a new window
+    3. Magnitude Calculation: Converts complex STFT output to magnitude spectrogram
+    4. Mel Conversion: Applies Mel filterbanks (128 filters in this case) to get Mel spectrogram
+    5. Logarithm: Takes the log of the Mel spectrogram (if `log=True`)
+    6. Normalization: Skips normalization since `normalize="NA"`
+    7. Padding: Pads the time dimension to a multiple of `pad_to` (default 16)    
+    """
+    previous_chunk = None
+    l_chunk_feat_seq_t = []
+    for chunk in chunks:
+        audio_signal_chunk = torch.tensor(chunk).unsqueeze(0).to(diar_model.device)
+        audio_signal_length_chunk = torch.tensor([audio_signal_chunk.shape[1]]).to(diar_model.device)
+        processed_signal_chunk, processed_signal_length_chunk = audio2mel.get_features(audio_signal_chunk, audio_signal_length_chunk)
+        if previous_chunk is not None:
+            to_add = previous_chunk[:, :, -99:]
+            total = torch.concat([to_add, processed_signal_chunk], dim=2)
+        else:
+            total = processed_signal_chunk
+        previous_chunk = processed_signal_chunk
+        l_chunk_feat_seq_t.append(torch.transpose(total, 1, 2))
+
+    batch_size = 1
+    streaming_state = init_streaming_state(diar_model.sortformer_modules,
+        batch_size = batch_size,
+        async_streaming = True,
+        device = diar_model.device
+    )
+    total_preds = torch.zeros((batch_size, 0, diar_model.sortformer_modules.n_spk), device=diar_model.device)
+
+    chunk_duration_seconds = diar_model.sortformer_modules.chunk_len * diar_model.sortformer_modules.subsampling_factor * diar_model.preprocessor._cfg.window_stride
+
+    l_speakers = [
+        {'start_time': 0,
+        'end_time': 0,
+        'speaker': 0
+        }
+    ]
+    len_prediction = None
+    left_offset = 0
+    right_offset = 8
+    for i, chunk_feat_seq_t in enumerate(l_chunk_feat_seq_t):
+        with torch.inference_mode():
+                streaming_state, total_preds = diar_model.forward_streaming_step(
+                    processed_signal=chunk_feat_seq_t,
+                    processed_signal_length=torch.tensor([chunk_feat_seq_t.shape[1]]),
+                    streaming_state=streaming_state,
+                    total_preds=total_preds,
+                    left_offset=left_offset,
+                    right_offset=right_offset,
+                )
+                left_offset = 8
+                preds_np = total_preds[0].cpu().numpy()
+                active_speakers = np.argmax(preds_np, axis=1)
+                if len_prediction is None:
+                    len_prediction = len(active_speakers) # we want to get the len of 1 prediction
+                frame_duration = chunk_duration_seconds / len_prediction
+                active_speakers = active_speakers[-len_prediction:]
+                for idx, spk in enumerate(active_speakers):
+                    if spk != l_speakers[-1]['speaker']:
+                        l_speakers.append(
+                            {'start_time': (i * chunk_duration_seconds + idx * frame_duration),
+                            'end_time': (i * chunk_duration_seconds + (idx + 1) * frame_duration),
+                            'speaker': spk
+                        })                    
+                    else:
+                        l_speakers[-1]['end_time'] = i * chunk_duration_seconds + (idx + 1) * frame_duration
+                    
+        
+        """
+        Should print
+        [{'start_time': 0, 'end_time': 8.72, 'speaker': 0}, 
+        {'start_time': 8.72, 'end_time': 18.88, 'speaker': 1},
+        {'start_time': 18.88, 'end_time': 24.96, 'speaker': 2},
+        {'start_time': 24.96, 'end_time': 31.68, 'speaker': 0}]
+        """
+    for speaker in l_speakers:
+        print(f"Speaker {speaker['speaker']}: {speaker['start_time']:.2f}s - {speaker['end_time']:.2f}s")    
+    
+
+if __name__ == '__main__':
+    
+    an4_audio = 'audio_test.mp3'
+    signal, sr = librosa.load(an4_audio, sr=16000)
+    signal = signal[:16000*30]
+    # signal = signal[:-(len(signal)%16000)]
+
+    print("\n" + "=" * 50)
+    print("Expected ground truth:")
+    print("Speaker 0: 0:00 - 0:09")
+    print("Speaker 1: 0:09 - 0:19") 
+    print("Speaker 2: 0:19 - 0:25")
+    print("Speaker 0: 0:25 - 0:30")
+    print("=" * 50)
+
+    chunk_size = 16000  # 1 second
+    chunks = []
+    for i in range(0, len(signal), chunk_size):
+        chunk = signal[i:i+chunk_size]
+        chunks.append(chunk)
+    
+    process_diarization(chunks)
--- a/whisperlivekit/ffmpeg_manager.py
+++ b/whisperlivekit/ffmpeg_manager.py
@@ -143,7 +143,7 @@ class FFmpegManager:
        try:
            data = await asyncio.wait_for(
                self.process.stdout.read(size),
-                timeout=5.0
+                timeout=20.0
            )
            return data
        except asyncio.TimeoutError:
--- a/whisperlivekit/parse_args.py
+++ b/whisperlivekit/parse_args.py
@@ -58,6 +58,14 @@ def parse_args():
        help="Hugging Face model ID for pyannote.audio embedding model.",
    )

+    parser.add_argument(
+        "--diarization-backend",
+        type=str,
+        default="sortformer",
+        choices=["sortformer", "diart"],
+        help="The diarization backend to use.",
+    )
+
    parser.add_argument(
        "--no-transcription",
        action="store_true",
@@ -74,7 +82,7 @@ def parse_args():
    parser.add_argument(
        "--model",
        type=str,
-        default="tiny",
+        default="small",
        help="Name size of the Whisper model to use (default: tiny). Suggested values: tiny.en,tiny,base.en,base,small.en,small,medium.en,medium,large-v1,large-v2,large-v3,large,large-v3-turbo. The model is automatically downloaded from the model hub if not present in model cache dir.",
    )
    
@@ -107,15 +115,15 @@ def parse_args():
    parser.add_argument(
        "--backend",
        type=str,
-        default="faster-whisper",
+        default="simulstreaming",
        choices=["faster-whisper", "whisper_timestamped", "mlx-whisper", "openai-api", "simulstreaming"],
        help="Load only this backend for Whisper processing.",
    )
    parser.add_argument(
-        "--vac",
+        "--no-vac",
        action="store_true",
        default=False,
-        help="Use VAC = voice activity controller. Recommended. Requires torch.",
+        help="Disable VAC = voice activity controller.",
    )
    parser.add_argument(
        "--vac-chunk-size", type=float, default=0.04, help="VAC sample size in seconds."
@@ -242,6 +250,14 @@ def parse_args():
        dest="model_path",
        help="Direct path to the SimulStreaming Whisper .pt model file. Overrides --model for SimulStreaming backend.",
    )
+    
+    simulstreaming_group.add_argument(
+        "--preloaded_model_count",
+        type=int,
+        default=1,
+        dest="preloaded_model_count",
+        help="Optional. Number of models to preload in memory to speed up loading (set up to the expected number of concurrent instances).",
+    )

    args = parser.parse_args()
    
--- a/whisperlivekit/remove_silences.py
+++ b/whisperlivekit/remove_silences.py
@@ -3,6 +3,7 @@ import re

 MIN_SILENCE_DURATION = 4 #in seconds
 END_SILENCE_DURATION = 8 #in seconds. you should keep it important to not have false positive when the model lag is important
+END_SILENCE_DURATION_VAC = 3 #VAC is good at detecting silences, but we want to skip the smallest silences

 def blank_to_silence(tokens):
    full_string = ''.join([t.text for t in tokens])
@@ -76,11 +77,15 @@ def no_token_to_silence(tokens):
            new_tokens.append(token)
    return new_tokens
            
-def ends_with_silence(tokens, current_time):
+def ends_with_silence(tokens, buffer_transcription, buffer_diarization, current_time, vac_detected_silence):
    if not tokens:
-        return []
+        return [], buffer_transcription, buffer_diarization
    last_token = tokens[-1]
-    if tokens and current_time - last_token.end >= END_SILENCE_DURATION:
+    if tokens and current_time and (
+        current_time - last_token.end >= END_SILENCE_DURATION 
+        or 
+        (current_time - last_token.end >= 3 and vac_detected_silence)
+        ):
        if last_token.speaker == -2:
            last_token.end = current_time
        else:
@@ -92,12 +97,14 @@ def ends_with_silence(tokens, current_time):
                    probability=0.95
                )
            )
-    return tokens
+        buffer_transcription = "" # for whisperstreaming backend, we should probably validate the buffer has because of the silence
+        buffer_diarization  = ""
+    return tokens, buffer_transcription, buffer_diarization
    

-def handle_silences(tokens, current_time):
+def handle_silences(tokens, buffer_transcription, buffer_diarization, current_time, vac_detected_silence):
    tokens = blank_to_silence(tokens) #useful for simulstreaming backend which tends to generate [BLANK_AUDIO] text
    tokens = no_token_to_silence(tokens)
-    tokens = ends_with_silence(tokens, current_time)
-    return tokens
+    tokens, buffer_transcription, buffer_diarization = ends_with_silence(tokens, buffer_transcription, buffer_diarization, current_time, vac_detected_silence)
+    return tokens, buffer_transcription, buffer_diarization
     
--- a/whisperlivekit/results_formater.py
+++ b/whisperlivekit/results_formater.py
@@ -0,0 +1,138 @@
+
+import logging
+from datetime import timedelta
+from whisperlivekit.remove_silences import handle_silences
+
+logger = logging.getLogger(__name__)
+logger.setLevel(logging.DEBUG)
+
+PUNCTUATION_MARKS = {'.', '!', '?'}
+CHECK_AROUND = 4
+
+def format_time(seconds: float) -> str:
+    """Format seconds as HH:MM:SS."""
+    return str(timedelta(seconds=int(seconds)))
+
+
+def is_punctuation(token):
+    if token.text.strip() in PUNCTUATION_MARKS:
+        return True
+    return False
+
+def next_punctuation_change(i, tokens):
+    for ind in range(i+1, min(len(tokens), i+CHECK_AROUND+1)):
+        if is_punctuation(tokens[ind]):
+            return ind        
+    return None
+
+def next_speaker_change(i, tokens, speaker):
+    for ind in range(i-1, max(0, i-CHECK_AROUND)-1, -1):
+        token = tokens[ind]
+        if is_punctuation(token):
+            break
+        if token.speaker != speaker:
+            return ind, token.speaker
+    return None, speaker
+    
+
+def new_line(
+    token,
+    speaker,
+    last_end_diarized,
+    debug_info = ""
+):
+    return {
+            "speaker": int(speaker),
+            "text": token.text + debug_info,
+            "beg": format_time(token.start),
+            "end": format_time(token.end),
+            "diff": round(token.end - last_end_diarized, 2)
+    }
+
+
+def append_token_to_last_line(lines, sep, token, debug_info, last_end_diarized):
+    if token.text:
+        lines[-1]["text"] += sep + token.text + debug_info
+        lines[-1]["end"] = format_time(token.end)
+        lines[-1]["diff"] = round(token.end - last_end_diarized, 2)
+            
+
+def format_output(state, silence, current_time, diarization, debug):
+    tokens = state["tokens"]
+    buffer_transcription = state["buffer_transcription"]
+    buffer_diarization = state["buffer_diarization"]
+    end_attributed_speaker = state["end_attributed_speaker"]
+    sep = state["sep"]
+    
+    previous_speaker = -1
+    lines = []
+    last_end_diarized = 0
+    undiarized_text = []
+    tokens, buffer_transcription, buffer_diarization = handle_silences(tokens, buffer_transcription, buffer_diarization, current_time, silence)
+    last_punctuation = None
+    for i, token in enumerate(tokens):
+        speaker = token.speaker
+        
+        if not diarization and speaker == -1: #Speaker -1 means no attributed by diarization. In the frontend, it should appear under 'Speaker 1'
+            speaker = 1
+        if diarization and not tokens[-1].speaker == -2:
+            if (speaker in [-1, 0]) and token.end >= end_attributed_speaker:
+                undiarized_text.append(token.text)
+                continue
+            elif (speaker in [-1, 0]) and token.end < end_attributed_speaker:
+                speaker = previous_speaker
+            if speaker not in [-1, 0]:
+                last_end_diarized = max(token.end, last_end_diarized)
+
+        debug_info = ""
+        if debug:
+            debug_info = f"[{format_time(token.start)} : {format_time(token.end)}]"
+            
+        if not lines:
+            lines.append(new_line(token, speaker, last_end_diarized, debug_info = ""))
+            continue
+        else:
+            previous_speaker = lines[-1]['speaker']
+        
+        if is_punctuation(token):
+            last_punctuation = i
+            
+        
+        if last_punctuation == i-1:
+            if speaker != previous_speaker:
+                # perfect, diarization perfectly aligned
+                lines.append(new_line(token, speaker, last_end_diarized, debug_info = ""))
+                last_punctuation, next_punctuation = None, None
+                continue
+            
+            speaker_change_pos, new_speaker = next_speaker_change(i, tokens, speaker)
+            if speaker_change_pos:
+                # Corrects delay:
+                # That was the idea. Okay haha |SPLIT SPEAKER| that's a good one 
+                # should become:
+                # That was the idea. |SPLIT SPEAKER| Okay haha that's a good one 
+                lines.append(new_line(token, new_speaker, last_end_diarized, debug_info = ""))
+            else:
+                # No speaker change to come
+                append_token_to_last_line(lines, sep, token, debug_info, last_end_diarized)
+            continue
+        
+
+        if speaker != previous_speaker:
+            if speaker == -2 or previous_speaker == -2: #silences can happen anytime
+                lines.append(new_line(token, speaker, last_end_diarized, debug_info = ""))
+                continue
+            elif next_punctuation_change(i, tokens):
+                # Corrects advance:
+                # Are you |SPLIT SPEAKER| okay? yeah, sure. Absolutely 
+                # should become:
+                # Are you okay? |SPLIT SPEAKER| yeah, sure. Absolutely 
+                append_token_to_last_line(lines, sep, token, debug_info, last_end_diarized)
+                continue
+            else: #we create a new speaker, but that's no ideal. We are not sure about the split. We prefer to append to previous line
+                # lines.append(new_line(token, speaker, last_end_diarized, debug_info = ""))
+                pass
+            
+        append_token_to_last_line(lines, sep, token, debug_info, last_end_diarized)
+    return lines, undiarized_text, buffer_transcription, '' 
+
--- a/whisperlivekit/whisper_streaming_custom/silero_vad_iterator.py
+++ b/whisperlivekit/whisper_streaming_custom/silero_vad_iterator.py
--- a/whisperlivekit/simul_whisper/backend.py
+++ b/whisperlivekit/simul_whisper/backend.py
@@ -4,9 +4,13 @@ import logging
 from typing import List, Tuple, Optional
 import logging
 from whisperlivekit.timed_objects import ASRToken, Transcript
+from whisperlivekit.warmup import load_file
 from whisperlivekit.simul_whisper.license_simulstreaming import SIMULSTREAMING_LICENSE
 from .whisper import load_model, tokenizer
+from .whisper.audio import TOKENS_PER_SECOND
+
 import os
+import gc
 logger = logging.getLogger(__name__)

 try:
@@ -19,6 +23,8 @@ except ImportError as e:
        """SimulStreaming dependencies are not available.
        Please install WhisperLiveKit using pip install "whisperlivekit[simulstreaming]".""")

+# TOO_MANY_REPETITIONS = 3
+
 class SimulStreamingOnlineProcessor:
    SAMPLING_RATE = 16000

@@ -30,33 +36,44 @@ class SimulStreamingOnlineProcessor:
    ):        
        self.asr = asr
        self.logfile = logfile
-        self.is_last = False
-        self.beg = 0.0
        self.end = 0.0
-        self.cumulative_audio_duration = 0.0
+        self.global_time_offset = 0.0
        
        self.committed: List[ASRToken] = []
-        self.last_result_tokens: List[ASRToken] = []        
-        self.model = PaddedAlignAttWhisper(
-            cfg=asr.cfg,
-            loaded_model=asr.whisper_model)
+        self.last_result_tokens: List[ASRToken] = []
+        self.load_new_backend()
+        
+        #can be moved
        if asr.tokenizer:
            self.model.tokenizer = asr.tokenizer

-    def insert_audio_chunk(self, audio: np.ndarray, audio_stream_end_time: Optional[float] = None):
+    def load_new_backend(self):
+        model = self.asr.get_new_model_instance()
+        self.model = PaddedAlignAttWhisper(
+            cfg=self.asr.cfg,
+            loaded_model=model)
+
+    def insert_silence(self, silence_duration, offset):
+        """
+        If silences are > 5s, we do a complete context clear. Otherwise, we just insert a small silence and shift the last_attend_frame
+        """
+        if silence_duration < 5:
+            gap_silence = torch.zeros(int(16000*silence_duration))
+            self.model.insert_audio(gap_silence)
+            # self.global_time_offset += silence_duration
+        else:
+            self.process_iter(is_last=True) #we want to totally process what remains in the buffer.
+            self.model.refresh_segment(complete=True)
+            self.global_time_offset += silence_duration + offset
+
+
+        
+    def insert_audio_chunk(self, audio: np.ndarray, audio_stream_end_time):
        """Append an audio chunk to be processed by SimulStreaming."""
            
        # Convert numpy array to torch tensor
        audio_tensor = torch.from_numpy(audio).float()
-        
-        # Update timing
-        chunk_duration = len(audio) / self.SAMPLING_RATE
-        self.cumulative_audio_duration += chunk_duration
-        
-        if audio_stream_end_time is not None:
-            self.end = audio_stream_end_time
-        else:
-            self.end = self.cumulative_audio_duration            
+        self.end = audio_stream_end_time #Only to be aligned with what happens in whisperstreaming backend.
        self.model.insert_audio(audio_tensor)

    def get_buffer(self):
@@ -68,38 +85,63 @@ class SimulStreamingOnlineProcessor:
        )

    def timestamped_text(self, tokens, generation):
-        # From the simulstreaming repo. self.model to self.asr.model
-        pr = generation["progress"]
-        if "result" not in generation:
-            split_words, split_tokens = self.model.tokenizer.split_to_word_tokens(tokens)
+        """
+        generate timestamped text from tokens and generation data.
+        
+        args:
+            tokens: List of tokens to process
+            generation: Dictionary containing generation progress and optionally results
+            
+        returns:
+            List of tuples containing (start_time, end_time, word) for each word
+        """
+        FRAME_DURATION = 0.02    
+        if "result" in generation:
+            split_words = generation["result"]["split_words"]
+            split_tokens = generation["result"]["split_tokens"]
        else:
-            split_words, split_tokens = generation["result"]["split_words"], generation["result"]["split_tokens"]
+            split_words, split_tokens = self.model.tokenizer.split_to_word_tokens(tokens)
+        progress = generation["progress"]
+        frames = [p["most_attended_frames"][0] for p in progress]
+        absolute_timestamps = [p["absolute_timestamps"][0] for p in progress]
+        tokens_queue = tokens.copy()
+        timestamped_words = []
+        
+        for word, word_tokens in zip(split_words, split_tokens):
+            # start_frame = None
+            # end_frame = None
+            for expected_token in word_tokens:
+                if not tokens_queue or not frames:
+                    raise ValueError(f"Insufficient tokens or frames for word '{word}'")
+                    
+                actual_token = tokens_queue.pop(0)
+                current_frame = frames.pop(0)
+                current_timestamp = absolute_timestamps.pop(0)
+                if actual_token != expected_token:
+                    raise ValueError(
+                        f"Token mismatch: expected '{expected_token}', "
+                        f"got '{actual_token}' at frame {current_frame}"
+                    )
+                # if start_frame is None:
+                #     start_frame = current_frame
+                # end_frame = current_frame
+            # start_time = start_frame * FRAME_DURATION
+            # end_time = end_frame * FRAME_DURATION
+            start_time = current_timestamp
+            end_time = current_timestamp + 0.1
+            timestamp_entry = (start_time, end_time, word)
+            timestamped_words.append(timestamp_entry)
+            logger.debug(f"TS-WORD:\t{start_time:.2f}\t{end_time:.2f}\t{word}")
+        return timestamped_words

-        frames = [p["most_attended_frames"][0] for p in pr]
-        tokens = tokens.copy()
-        ret = []
-        for sw,st in zip(split_words,split_tokens):
-            b = None
-            for stt in st:
-                t,f = tokens.pop(0), frames.pop(0)
-                if t != stt:
-                    raise ValueError(f"Token mismatch: {t} != {stt} at frame {f}.")
-                if b is None:
-                    b = f
-            e = f
-            out = (b*0.02, e*0.02, sw)
-            ret.append(out)
-            logger.debug(f"TS-WORD:\t{' '.join(map(str, out))}")
-        return ret
-
-    def process_iter(self) -> Tuple[List[ASRToken], float]:
+    def process_iter(self, is_last=False) -> Tuple[List[ASRToken], float]:
        """
        Process accumulated audio chunks using SimulStreaming.
        
        Returns a tuple: (list of committed ASRToken objects, float representing the audio processed up to time).
        """
-        try:            
-            tokens, generation_progress = self.model.infer(is_last=self.is_last)
+        try:
+            tokens, generation_progress = self.model.infer(is_last=is_last)
            ts_words = self.timestamped_text(tokens, generation_progress)
            
            new_tokens = []
@@ -111,9 +153,33 @@ class SimulStreamingOnlineProcessor:
                    end=end,
                    text=word,
                    probability=0.95  # fake prob. Maybe we can extract it from the model?
+                ).with_offset(
+                    self.global_time_offset
                )
                new_tokens.append(token)
-                self.committed.extend(new_tokens)
+                
+            # identical_tokens = 0
+            # n_new_tokens = len(new_tokens)
+            # if n_new_tokens:
+            
+            self.committed.extend(new_tokens)
+            
+            # if token in self.committed:
+            #     pos = len(self.committed) - 1 - self.committed[::-1].index(token)
+            # if pos:
+            #     for i in range(len(self.committed) - n_new_tokens, -1, -n_new_tokens):
+            #         commited_segment = self.committed[i:i+n_new_tokens]
+            #         if commited_segment == new_tokens:
+            #             identical_segments +=1
+            #             if identical_tokens >= TOO_MANY_REPETITIONS:
+            #                 logger.warning('Too many repetition, model is stuck. Load a new one')
+            #                 self.committed = self.committed[:i]
+            #                 self.load_new_backend()
+            #                 return [], self.end
+
+            # pos = self.committed.rindex(token)
+
+            
            
            return new_tokens, self.end

@@ -132,6 +198,13 @@ class SimulStreamingOnlineProcessor:
        except Exception as e:
            logger.exception(f"SimulStreaming warmup failed: {e}")

+    def __del__(self):
+        # free the model and add a new model to stack.
+        # del self.model
+        gc.collect()
+        torch.cuda.empty_cache()
+        # self.asr.new_model_to_stack()
+        self.model.remove_hooks()

 class SimulStreamingASR():
    """SimulStreaming backend with AlignAtt policy."""
@@ -141,11 +214,11 @@ class SimulStreamingASR():
        logger.warning(SIMULSTREAMING_LICENSE)
        self.logfile = logfile
        self.transcribe_kargs = {}
-        self.original_language = None if lan == "auto" else lan
+        self.original_language = lan
        
        self.model_path = kwargs.get('model_path', './large-v3.pt')
        self.frame_threshold = kwargs.get('frame_threshold', 25)
-        self.audio_max_len = kwargs.get('audio_max_len', 30.0)
+        self.audio_max_len = kwargs.get('audio_max_len', 20.0)
        self.audio_min_len = kwargs.get('audio_min_len', 0.0)
        self.segment_length = kwargs.get('segment_length', 0.5)
        self.beams = kwargs.get('beams', 1)
@@ -156,6 +229,8 @@ class SimulStreamingASR():
        self.init_prompt = kwargs.get('init_prompt', None)
        self.static_init_prompt = kwargs.get('static_init_prompt', None)
        self.max_context_tokens = kwargs.get('max_context_tokens', None)
+        self.warmup_file = kwargs.get('warmup_file', None)
+        self.preload_model_count = kwargs.get('preload_model_count', 1)
        
        if model_dir is not None:
            self.model_path = model_dir
@@ -176,16 +251,6 @@ class SimulStreamingASR():
            }
            self.model_path = model_mapping.get(modelsize, f'./{modelsize}.pt')
        
-        self.model = self.load_model(modelsize)
-        
-        # Set up tokenizer for translation if needed
-        if self.task == "translate":
-            self.tokenizer = self.set_translate_task()
-        else:
-            self.tokenizer = None
-
-
-    def load_model(self, modelsize):
        self.cfg = AlignAttConfig(
                model_path=self.model_path,
                segment_length=self.segment_length,
@@ -201,23 +266,55 @@ class SimulStreamingASR():
                init_prompt=self.init_prompt,
                max_context_tokens=self.max_context_tokens,
                static_init_prompt=self.static_init_prompt,
-        )   
-        model_name = os.path.basename(self.cfg.model_path).replace(".pt", "")
-        model_path = os.path.dirname(os.path.abspath(self.cfg.model_path))
-        self.whisper_model = load_model(name=model_name, download_root=model_path)
+        )  
+        
+        # Set up tokenizer for translation if needed
+        if self.task == "translate":
+            self.tokenizer = self.set_translate_task()
+        else:
+            self.tokenizer = None
+        
+        self.model_name = os.path.basename(self.cfg.model_path).replace(".pt", "")
+        self.model_path = os.path.dirname(os.path.abspath(self.cfg.model_path))
+        self.models = [self.load_model() for i in range(self.preload_model_count)]
+    
+
+
+
+    def load_model(self):
+        whisper_model = load_model(name=self.model_name, download_root=self.model_path)
+        warmup_audio = load_file(self.warmup_file)
+        whisper_model.transcribe(warmup_audio, language=self.original_language if self.original_language != 'auto' else None)
+        return whisper_model
+    
+    def get_new_model_instance(self):
+        """
+        SimulStreaming cannot share the same backend because it uses global forward hooks on the attention layers.
+        Therefore, each user requires a separate model instance, which can be memory-intensive. To maintain speed, we preload the models into memory.
+        """
+        if len(self.models) == 0:
+            self.models.append(self.load_model())
+        new_model = self.models.pop()
+        return new_model
+        # self.models[0]
+
+    def new_model_to_stack(self):
+        self.models.append(self.load_model())
        

    def set_translate_task(self):
        """Set up translation task."""
+        if self.cfg.language == 'auto':
+            raise Exception('Translation cannot be done with language = auto')
        return tokenizer.get_tokenizer(
            multilingual=True,
-            language=self.model.cfg.language,
-            num_languages=self.model.model.num_languages,
+            language=self.cfg.language,
+            num_languages=99,
            task="translate"
        )

    def transcribe(self, audio):
        """
-        Only used for warmup. It's a direct whisper call, not a simulstreaming call
+        Warmup is done directly in load_model
        """
-        self.whisper_model.transcribe(audio, language=self.original_language)
+        pass
--- a/whisperlivekit/simul_whisper/config.py
+++ b/whisperlivekit/simul_whisper/config.py
@@ -24,6 +24,6 @@ class AlignAttConfig(SimulWhisperConfig):
    segment_length: float = field(default=1.0, metadata = {"help": "in second"})
    frame_threshold: int = 4
    rewind_threshold: int = 200
-    audio_max_len: float = 30.0
+    audio_max_len: float = 20.0
    cif_ckpt_path: str = ""
    never_fire: bool = False
--- a/whisperlivekit/simul_whisper/simul_whisper.py
+++ b/whisperlivekit/simul_whisper/simul_whisper.py
@@ -56,6 +56,7 @@ class PaddedAlignAttWhisper:
        self.max_text_len = self.model.dims.n_text_ctx
        self.num_decoder_layers = len(self.model.decoder.blocks)
        self.cfg = cfg
+        self.l_hooks = []

        # model to detect end-of-word boundary at the end of the segment
        self.CIFLinear, self.always_fire, self.never_fire = load_cif(cfg,
@@ -69,7 +70,8 @@ class PaddedAlignAttWhisper:
            t = F.softmax(net_output[1], dim=-1)
            self.dec_attns.append(t.squeeze(0))
        for b in self.model.decoder.blocks:
-            b.cross_attn.register_forward_hook(layer_hook)
+            hook = b.cross_attn.register_forward_hook(layer_hook)
+            self.l_hooks.append(hook)
        
        self.kv_cache = {}
        def kv_hook(module: torch.nn.Linear, _, net_output: torch.Tensor):
@@ -82,10 +84,13 @@ class PaddedAlignAttWhisper:
            return self.kv_cache[module.cache_id] 

        for i,b in enumerate(self.model.decoder.blocks):
-            b.attn.key.register_forward_hook(kv_hook)
-            b.attn.value.register_forward_hook(kv_hook)
-            b.cross_attn.key.register_forward_hook(kv_hook)
-            b.cross_attn.value.register_forward_hook(kv_hook)
+            hooks = [
+                b.attn.key.register_forward_hook(kv_hook),
+                b.attn.value.register_forward_hook(kv_hook),
+                b.cross_attn.key.register_forward_hook(kv_hook),
+                b.cross_attn.value.register_forward_hook(kv_hook),
+            ]
+            self.l_hooks.extend(hooks)

        self.align_source = {}
        self.num_align_heads = 0
@@ -120,6 +125,7 @@ class PaddedAlignAttWhisper:
        self.init_tokens()
        
        self.last_attend_frame = -self.cfg.rewind_threshold
+        self.cumulative_time_offset = 0.0

        if self.cfg.max_context_tokens is None:
            self.max_context_tokens = self.max_text_len
@@ -139,6 +145,11 @@ class PaddedAlignAttWhisper:
            self.inference.kv_cache = self.kv_cache

            self.token_decoder = BeamSearchDecoder(inference=self.inference, eot=self.tokenizer.eot, beam_size=cfg.beam_size)
+            
+    def remove_hooks(self):
+        print('remove hook')
+        for hook in self.l_hooks:
+            hook.remove()

    def create_tokenizer(self, language=None):
        self.tokenizer = tokenizer.get_tokenizer(
@@ -210,6 +221,7 @@ class PaddedAlignAttWhisper:
        self.init_tokens()
        self.last_attend_frame = -self.cfg.rewind_threshold       
        self.detected_language = None
+        self.cumulative_time_offset = 0.0
        self.init_context()
        logger.debug(f"Context: {self.context}")
        if not complete and len(self.segments) > 2:
@@ -277,8 +289,9 @@ class PaddedAlignAttWhisper:
            removed_len = self.segments[0].shape[0] / 16000
            segments_len -= removed_len
            self.last_attend_frame -= int(TOKENS_PER_SECOND*removed_len)
+            self.cumulative_time_offset += removed_len  # Track cumulative time removed
            self.segments = self.segments[1:]
-            logger.debug(f"remove segments: {len(self.segments)} {len(self.tokens)}")
+            logger.debug(f"remove segments: {len(self.segments)} {len(self.tokens)}, cumulative offset: {self.cumulative_time_offset:.2f}s")
            if len(self.tokens) > 1:
                self.context.append_token_ids(self.tokens[1][0,:])
                self.tokens = [self.initial_tokens] + self.tokens[2:]
@@ -494,7 +507,13 @@ class PaddedAlignAttWhisper:
            # for each beam, the most attended frame is:
            most_attended_frames = torch.argmax(attn_of_alignment_heads[:,-1,:], dim=-1)
            generation_progress_loop.append(("most_attended_frames",most_attended_frames.clone().tolist()))
+            
+            # Calculate absolute timestamps accounting for cumulative offset
+            absolute_timestamps = [(frame * 0.02 + self.cumulative_time_offset) for frame in most_attended_frames.tolist()]
+            generation_progress_loop.append(("absolute_timestamps", absolute_timestamps))
+            
            logger.debug(str(most_attended_frames.tolist()) + " most att frames")
+            logger.debug(f"Absolute timestamps: {absolute_timestamps} (offset: {self.cumulative_time_offset:.2f}s)")

            most_attended_frame = most_attended_frames[0].item()

@@ -599,4 +618,4 @@ class PaddedAlignAttWhisper:
        
        self._clean_cache()

-        return new_hypothesis, generation
+        return new_hypothesis, generation
--- a/whisperlivekit/timed_objects.py
+++ b/whisperlivekit/timed_objects.py
@@ -29,4 +29,8 @@ class SpeakerSegment(TimedText):
    """Represents a segment of audio attributed to a specific speaker.
    No text nor probability is associated with this segment.
    """
-    pass
+    pass
+
+@dataclass
+class Silence():
+    duration: float
--- a/whisperlivekit/trail_repetition.py
+++ b/whisperlivekit/trail_repetition.py
@@ -0,0 +1,60 @@
+from typing import Sequence, Callable, Any, Optional, Dict
+
+def _detect_tail_repetition(
+    seq: Sequence[Any],
+    key: Callable[[Any], Any] = lambda x: x,  # extract comparable value
+    min_block: int = 1,                       # set to 2 to ignore 1-token loops like "."
+    max_tail: int = 300,                      # search window from the end for speed
+    prefer: str = "longest",                  # "longest" coverage or "smallest" block
+) -> Optional[Dict]:
+    vals = [key(x) for x in seq][-max_tail:]
+    n = len(vals)
+    best = None
+
+    # try every possible block length
+    for b in range(min_block, n // 2 + 1):
+        block = vals[-b:]
+        # count how many times this block repeats contiguously at the very end
+        count, i = 0, n
+        while i - b >= 0 and vals[i - b:i] == block:
+            count += 1
+            i -= b
+
+        if count >= 2:
+            cand = {
+                "block_size": b,
+                "count": count,
+                "start_index": len(seq) - count * b,  # in original seq
+                "end_index": len(seq),
+            }
+            if (best is None or
+                (prefer == "longest" and count * b > best["count"] * best["block_size"]) or
+                (prefer == "smallest" and b < best["block_size"])):
+                best = cand
+    return best
+
+def trim_tail_repetition(
+    seq: Sequence[Any],
+    key: Callable[[Any], Any] = lambda x: x,
+    min_block: int = 1,
+    max_tail: int = 300,
+    prefer: str = "longest",
+    keep: int = 1,  # how many copies of the repeating block to keep at the end (0 or 1 are common)
+):
+    """
+    Returns a new sequence with repeated tail trimmed.
+    keep=1 -> keep a single copy of the repeated block.
+    keep=0 -> remove all copies of the repeated block.
+    """
+    rep = _detect_tail_repetition(seq, key, min_block, max_tail, prefer)
+    if not rep:
+        return seq, False  # nothing to trim
+
+    b, c = rep["block_size"], rep["count"]
+    if keep < 0:
+        keep = 0
+    if keep >= c:
+        return seq, False  # nothing to trim (already <= keep copies)
+    # new length = total - (copies_to_remove * block_size)
+    new_len = len(seq) - (c - keep) * b
+    return seq[:new_len], True
--- a/whisperlivekit/translate/gemma_translate.py
+++ b/whisperlivekit/translate/gemma_translate.py
@@ -0,0 +1,60 @@
+# gemma_translate.py
+import argparse
+import torch
+from transformers import AutoTokenizer, AutoModelForCausalLM
+
+MODEL_ID = "google/gemma-3-270m-it"
+
+def build_prompt(tokenizer, text, target_lang, source_lang=None):
+    # Use the model's chat template for best results
+    if source_lang:
+        user_msg = (
+            f"Translate the following {source_lang} text into {target_lang}.\n"
+            f"Return only the translation.\n\n"
+            f"Text:\n{text}"
+        )
+    else:
+        user_msg = (
+            f"Translate the following text into {target_lang}.\n"
+            f"Return only the translation.\n\n"
+            f"Text:\n{text}"
+        )
+    chat = [{"role": "user", "content": user_msg}]
+    return tokenizer.apply_chat_template(chat, tokenize=False, add_generation_prompt=True)
+
+def translate(text, target_lang, source_lang=None, max_new_tokens=256, temperature=0.2, top_p=0.95):
+    tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
+    model = AutoModelForCausalLM.from_pretrained(
+        MODEL_ID,
+        torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32,
+        device_map="auto"
+    )
+
+    prompt = build_prompt(tokenizer, text, target_lang, source_lang)
+    inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
+
+    with torch.no_grad():
+        output_ids = model.generate(
+            **inputs,
+            max_new_tokens=max_new_tokens,
+            temperature=temperature,
+            top_p=top_p,
+            do_sample=temperature > 0.0,
+            eos_token_id=tokenizer.eos_token_id,
+        )
+
+    # Slice off the prompt to keep only the assistant answer
+    generated_ids = output_ids[0][inputs["input_ids"].shape[1]:]
+    out = tokenizer.decode(generated_ids, skip_special_tokens=True).strip()
+    return out
+
+if __name__ == "__main__":
+    ap = argparse.ArgumentParser(description="Translate with google/gemma-3-270m-it")
+    ap.add_argument("--text", required=True, help="Text to translate")
+    ap.add_argument("--to", dest="target_lang", required=True, help="Target language (e.g., French, Spanish)")
+    ap.add_argument("--from", dest="source_lang", default=None, help="Source language (optional)")
+    ap.add_argument("--temp", type=float, default=0.2, help="Sampling temperature (0 = deterministic-ish)")
+    ap.add_argument("--max-new", type=int, default=256, help="Max new tokens")
+    args = ap.parse_args()
+
+    print(translate(args.text, args.target_lang, args.source_lang, max_new_tokens=args.max_new, temperature=args.temp))
--- a/whisperlivekit/translate/nllb_translate.py
+++ b/whisperlivekit/translate/nllb_translate.py
@@ -0,0 +1,121 @@
+# nllb_translate.py
+import argparse
+from pathlib import Path
+from typing import List
+import torch
+from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
+
+MODEL_ID = "facebook/nllb-200-distilled-600M"
+
+# Common language shortcuts → NLLB codes (extend as needed)
+LANG_MAP = {
+    "english": "eng_Latn",
+    "en": "eng_Latn",
+    "french": "fra_Latn",
+    "fr": "fra_Latn",
+    "spanish": "spa_Latn",
+    "es": "spa_Latn",
+    "german": "deu_Latn",
+    "de": "deu_Latn",
+    "italian": "ita_Latn",
+    "it": "ita_Latn",
+    "portuguese": "por_Latn",
+    "pt": "por_Latn",
+    "arabic": "arb_Arab",
+    "ar": "arb_Arab",
+    "russian": "rus_Cyrl",
+    "ru": "rus_Cyrl",
+    "turkish": "tur_Latn",
+    "tr": "tur_Latn",
+    "chinese": "zho_Hans",
+    "zh": "zho_Hans",           # Simplified
+    "zh-cn": "zho_Hans",
+    "zh-hans": "zho_Hans",
+    "zh-hant": "zho_Hant",      # Traditional
+    "japanese": "jpn_Jpan",
+    "ja": "jpn_Jpan",
+    "korean": "kor_Hang",
+    "ko": "kor_Hang",
+    "dutch": "nld_Latn",
+    "nl": "nld_Latn",
+    "polish": "pol_Latn",
+    "pl": "pol_Latn",
+    "swedish": "swe_Latn",
+    "sv": "swe_Latn",
+    "norwegian": "nob_Latn",
+    "no": "nob_Latn",
+    "danish": "dan_Latn",
+    "da": "dan_Latn",
+    "finnish": "fin_Latn",
+    "fi": "fin_Latn",
+    "catalan": "cat_Latn",
+    "ca": "cat_Latn",
+    "hindi": "hin_Deva",
+    "hi": "hin_Deva",
+    "vietnamese": "vie_Latn",
+    "vi": "vie_Latn",
+    "indonesian": "ind_Latn",
+    "id": "ind_Latn",
+    "thai": "tha_Thai",
+    "th": "tha_Thai",
+}
+
+def norm_lang(code: str) -> str:
+    c = code.strip().lower()
+    return LANG_MAP.get(c, code)
+
+def translate_texts(texts: List[str], src_code: str, tgt_code: str,
+                    max_new_tokens=512, device=None, dtype=None) -> List[str]:
+    tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, src_lang=src_code)
+    model = AutoModelForSeq2SeqLM.from_pretrained(
+        MODEL_ID,
+        torch_dtype=dtype if dtype is not None else (torch.float16 if torch.cuda.is_available() else torch.float32),
+        device_map="auto" if torch.cuda.is_available() else None,
+    )
+    if device:
+        model.to(device)
+
+    inputs = tokenizer(texts, return_tensors="pt", padding=True, truncation=True)
+    if device or torch.cuda.is_available():
+        inputs = {k: v.to(model.device) for k, v in inputs.items()}
+
+    forced_bos = tokenizer.convert_tokens_to_ids(tgt_code)
+    with torch.no_grad():
+        gen = model.generate(
+            **inputs,
+            max_new_tokens=max_new_tokens,
+            forced_bos_token_id=forced_bos,
+        )
+    outs = tokenizer.batch_decode(gen, skip_special_tokens=True)
+    return [o.strip() for o in outs]
+
+def main():
+    ap = argparse.ArgumentParser(description="Translate with facebook/nllb-200-distilled-600M")
+    ap.add_argument("--text", help="Inline text to translate")
+    ap.add_argument("--file", help="Path to a UTF-8 text file (one example per line)")
+    ap.add_argument("--src", required=True, help="Source language (e.g. fr, fra_Latn)")
+    ap.add_argument("--tgt", required=True, help="Target language (e.g. en, eng_Latn)")
+    ap.add_argument("--max-new", type=int, default=512, help="Max new tokens")
+    args = ap.parse_args()
+
+    src = norm_lang(args.src)
+    tgt = norm_lang(args.tgt)
+
+    batch: List[str] = []
+    if args.text:
+        batch.append(args.text)
+    if args.file:
+        lines = Path(args.file).read_text(encoding="utf-8").splitlines()
+        batch.extend([ln for ln in lines if ln.strip()])
+
+    if not batch:
+        raise SystemExit("Provide --text or --file")
+
+    results = translate_texts(batch, src, tgt, max_new_tokens=args.max_new)
+    for i, (inp, out) in enumerate(zip(batch, results), 1):
+        print(f"\n--- Sample {i} ---")
+        print(f"SRC [{src}]: {inp}")
+        print(f"TGT [{tgt}]: {out}")
+
+if __name__ == "__main__":
+    main()
--- a/whisperlivekit/translate/sentence_segmenter.py
+++ b/whisperlivekit/translate/sentence_segmenter.py
@@ -0,0 +1,38 @@
+import regex
+from functools import lru_cache
+class SentenceSegmenter:
+
+    """
+    Regex sentence splitter for Latin languages, Japanese and Chinese.
+    It is based on sacrebleu TokenizerV14International(BaseTokenizer).
+    
+    Returns: a list of strings, where each string is a sentence.
+    Spaces following punctuation are appended after punctuation within the sequence.
+    Total number of characters in the output is the same as in the input.  
+    """
+
+    sep = 'ŽžŽžSentenceSeparatorŽžŽž'  # string that certainly won't be in src or target
+    latin_terminals = '!?.'
+    jap_zh_terminals = '。！？'
+    terminals = latin_terminals + jap_zh_terminals
+
+    def __init__(self):
+        # end of sentence characters:
+        terminals = self.terminals
+        self._re = [
+            # Separate out punctuations preceeded by a non-digit. 
+            # If followed by space-like sequence of characters, they are 
+            # appended to the punctuation, not to the next sequence.
+            (regex.compile(r'(\P{N})(['+terminals+r'])(\p{Z}*)'), r'\1\2\3'+self.sep),
+            # Separate out punctuations followed by a non-digit
+            (regex.compile(r'('+terminals+r')(\P{N})'), r'\1'+self.sep+r'\2'),
+#            # Separate out symbols
+            # -> no, we don't tokenize but segment the punctuation
+#            (regex.compile(r'(\p{S})'), r' \1 '),
+        ]
+
+    @lru_cache(maxsize=2**16)
+    def __call__(self, line):
+        for (_re, repl) in self._re:
+            line = _re.sub(repl, line)
+        return [ t for t in line.split(self.sep) if t != '' ]
--- a/whisperlivekit/translate/simul_llm_translate.py
+++ b/whisperlivekit/translate/simul_llm_translate.py
@@ -0,0 +1,466 @@
+import sys
+
+import ctranslate2
+import sentencepiece as spm
+import transformers
+import argparse
+
+def generate_words(sp, step_results):
+    tokens_buffer = []
+
+    for step_result in step_results:
+        is_new_word = step_result.token.startswith("▁")
+
+        if is_new_word and tokens_buffer:
+            word = sp.decode(tokens_buffer)
+            if word:
+                yield word
+            tokens_buffer = []
+
+        tokens_buffer.append(step_result.token_id)
+
+    if tokens_buffer:
+        word = sp.decode(tokens_buffer)
+        if word:
+            yield word
+
+from sentence_segmenter import SentenceSegmenter
+
+class LLMTranslator:
+
+    def __init__(self, system_prompt='Please translate.', max_context_length=4096, len_ratio=None):
+        self.system_prompt = system_prompt
+
+
+        print("Loading the model...", file=sys.stderr)
+        self.generator = ctranslate2.Generator("ct2_EuroLLM-9B-Instruct/", device="cuda")
+        self.sp = spm.SentencePieceProcessor("EuroLLM-9B-Instruct/tokenizer.model")
+        self.tokenizer = transformers.AutoTokenizer.from_pretrained("EuroLLM-9B-Instruct/")
+        print("...done", file=sys.stderr)
+
+        self.max_context_length = max_context_length
+
+        self.max_tokens_to_trim = self.max_context_length - 10
+        self.len_ratio = len_ratio
+
+        # my regex sentence segmenter
+        self.segmenter = SentenceSegmenter()
+
+#        self.max_generation_length = 512
+#        self.max_prompt_length = context_length - max_generation_length
+
+    def start_dialog(self):
+        return [{'role':'system', 'content': self.system_prompt }]
+    
+
+    def build_prompt(self, dialog):
+        toks = self.tokenizer.apply_chat_template(dialog, tokenize=True, add_generation_prompt=False)
+        if len(dialog) == 3:
+            toks = toks[:-2]
+        print("len toks:", len(toks), file=sys.stderr)
+#        print(toks, file=sys.stderr)
+
+        c = self.tokenizer.convert_ids_to_tokens(toks)
+#        print(c,file=sys.stderr)
+        return c
+
+    def translate(self, src, tgt_forced=""):
+        #src, tgt_forced = self.trim(src, tgt_forced)
+
+        dialog = self.start_dialog()
+        dialog += [{'role':'user','content': src}]
+        if tgt_forced != "":
+            dialog += [{'role':'assistant','content': tgt_forced}]
+
+        prompt_tokens = self.build_prompt(dialog)
+        if self.len_ratio is not None:
+            limit_len = int(len(self.tokenizer.encode(src)) * self.len_ratio) + 10
+            limit_kw = {'max_length': limit_len}
+        else:
+            limit_kw = {}
+        step_results = self.generator.generate_tokens(
+            prompt_tokens,
+            **limit_kw, 
+    #    end_token=tokenizer.eos_token,
+    #            sampling_temperature=0.6,
+    #            sampling_topk=20,
+    #            sampling_topp=1,
+        )
+
+        res = []
+        #output_ids = []
+        for step_result in step_results:
+        #    is_new_word = step_result.token.startswith("▁")
+        #    if is_new_word and output_ids:
+        #        word = self.sp.decode(output_ids)
+#                print(word, end=" ", flush=True, file=sys.stderr)
+        #        output_ids = []
+        #    output_ids.append(step_result.token_id)
+            res.append(step_result)
+
+        #if output_ids:
+        #    word = self.sp.decode(output_ids)
+#        print(word, file=sys.stderr)
+
+        return self.sp.decode([r.token_id for r in res])
+ #       print(res)
+ #       print([s.token for s in res], file=sys.stderr)
+#        print([s.token==self.tokenizer.eos_token for s in res], file=sys.stderr)
+
+class ParallelTextBuffer:
+    def __init__(self, tokenizer, max_tokens, trimming="segments", init_src="", init_tgt=""):
+        self.tokenizer = tokenizer
+        self.max_tokens = max_tokens
+
+        self.src_buffer = []  # list of lists
+        if init_src:
+            self.src_buffer.append(init_src)
+
+        self.tgt_buffer = []  # list of strings
+        if init_tgt:
+            self.tgt_buffer.append(init_tgt)
+
+        self.trimming = trimming
+        if self.trimming == "sentences":
+            self.segmenter = SentenceSegmenter()
+
+    def len_src(self):
+        return sum(len(t) for t in self.src_buffer) + len(self.src_buffer) - 1
+
+    def insert(self, src, tgt):
+        self.src_buffer.append(src)
+        self.tgt_buffer.append(tgt)
+
+    def insert_src_suffix(self, s):
+        if self.src_buffer:
+            self.src_buffer[-1][-1] += s
+        else:
+            self.src_buffer.append([s])
+
+    def trim_sentences(self):
+        # src_tok_lens = [len(self.tokenizer.encode(" ".join(b))) for b in self.src_buffer]
+        # tgt_tok_lens = [len(self.tokenizer.encode(t)) for t in self.tgt_buffer]
+
+        src = " ".join(" ".join(b) for b in self.src_buffer)
+        tgt = "".join(self.tgt_buffer)
+
+        src_sp_toks = self.tokenizer.encode(src)
+        tgt_sp_toks = self.tokenizer.encode(tgt)
+
+
+
+        def trim_sentence(text):
+            sents = self.segmenter(text)
+            print("SENTS:", len(sents), sents, file=sys.stderr)
+            return "".join(sents[1:])
+
+        while len(src_sp_toks) + len(tgt_sp_toks) > self.max_tokens:
+            nsrc = trim_sentence(src)
+            ntgt = trim_sentence(tgt)
+            if not nsrc or not ntgt:
+                print("src or tgt is empty after trimming.", file=sys.stderr)
+                print("src: ", src, file=sys.stderr)
+                print("tgt: ", tgt, file=sys.stderr)
+                break
+            src = nsrc
+            tgt = ntgt
+            src_sp_toks = self.tokenizer.encode(src)
+            tgt_sp_toks = self.tokenizer.encode(tgt)
+            print("TRIMMED SRC:", (src,), file=sys.stderr)
+            print("TRIMMED TGT:", (tgt,), file=sys.stderr)
+
+        self.src_buffer = [src.split()]
+        self.tgt_buffer = [tgt]
+        return src, tgt
+
+    def trim_segments(self):
+        print("BUFFER:", file=sys.stderr)
+        for s,t in zip(self.src_buffer, self.tgt_buffer):
+            print("\t", s,"...",t,file=sys.stderr) #,self.src_buffer, self.tgt_buffer, file=sys.stderr)
+        src = " ".join(" ".join(b) for b in self.src_buffer)
+        tgt = "".join(self.tgt_buffer)
+
+        src_sp_toks = self.tokenizer.encode(src)
+        tgt_sp_toks = self.tokenizer.encode(tgt)
+
+        while len(src_sp_toks) + len(tgt_sp_toks) > self.max_tokens:
+            if len(self.src_buffer) > 1 and len(self.tgt_buffer) > 1:
+                self.src_buffer.pop(0)
+                self.tgt_buffer.pop(0)
+            else:
+                break
+            src = " ".join(" ".join(b) for b in self.src_buffer)
+            tgt = "".join(self.tgt_buffer)
+
+            src_sp_toks = self.tokenizer.encode(src)
+            tgt_sp_toks = self.tokenizer.encode(tgt)
+
+        print("TRIMMED SEGMENTS SRC:", (src,), file=sys.stderr)
+        print("TRIMMED SEGMENTS TGT:", (tgt,), file=sys.stderr)
+
+        return src, tgt
+
+    def trim(self):
+        if self.trimming == "sentences":
+            return self.trim_sentences()
+        return self.trim_segments()
+
+
+
+class SimulLLM:
+
+    def __init__(self, llmtrans, min_len=0, chunk=1, trimming="sentences", language="ja", init_src="", init_tgt=""):
+        self.llmtranslator = llmtrans
+
+        #self.src_buffer = init_src 
+        #self.confirmed_tgt = init_tgt
+
+        self.buffer = ParallelTextBuffer(self.llmtranslator.tokenizer, self.llmtranslator.max_tokens_to_trim, trimming=trimming, init_src=init_src, init_tgt=init_tgt)
+
+        self.last_inserted = []
+        self.last_unconfirmed = ""
+
+        self.min_len = min_len
+
+        self.step = chunk 
+        self.language = language
+        if language in ["ja", "zh"]:
+            self.specific_space = ""
+        else:
+            self.specific_space = " "
+
+    def insert(self, src):
+        if isinstance(src, str):
+            self.last_inserted.append(src)
+        else:
+            self.last_inserted += src
+
+    def insert_suffix(self, text):
+        '''
+        Insert suffix of a word to the last inserted word. 
+        It may be because the word was split to multiple parts in the input, each with different timestamps.
+        '''
+        if self.last_inserted:
+            self.last_inserted[-1] += text
+        elif self.src_buffer:
+            self.buffer.insert_src_suffix(text)
+        else:
+            # this shouldn't happen
+            self.last_inserted.append(text)
+
+    def trim_longest_common_prefix(self, a,b):
+        if self.language not in ["ja", "zh"]:
+            a = a.split()
+            b = b.split()
+        i = 0
+        for i,(x,y) in enumerate(zip(a,b)):
+            if x != y:
+                break
+        if self.language in ["ja", "zh"]:
+            #print("tady160",(a, b, i), file=sys.stderr)
+            return a[:i], b[i:]
+        else:
+            return " ".join(a[:i]), " ".join(b[i:])
+
+    def process_iter(self):
+        if self.buffer.len_src() + len(self.last_inserted) < self.min_len:
+            return ""
+
+        src, forced_tgt = self.buffer.trim() #llmtranslator.trim(" ".join(self.src_buffer), self.confirmed_tgt)
+        #self.src_buffer = self.src_buffer.split()
+        #src = " ".join(self.src_buffer)
+
+        confirmed_out = ""
+        run = False
+        for i in range(self.step, len(self.last_inserted), self.step):
+            for w in self.last_inserted[i-self.step:i]:
+                src += " " + w
+                run = True
+            if not run: break
+            
+            print("SRC",src,file=sys.stderr)
+
+            print("FORCED TGT",forced_tgt,file=sys.stderr)
+            out = self.llmtranslator.translate(src, forced_tgt)
+            print("OUT",out,file=sys.stderr)
+            confirmed, unconfirmed = self.trim_longest_common_prefix(self.last_unconfirmed, out)
+            self.last_unconfirmed = unconfirmed
+            #print("tady", (self.confirmed_tgt, self.specific_space, confirmed), file=sys.stderr)
+            if confirmed:
+#                self.confirmed_tgt += self.specific_space + confirmed
+#            print(confirmed_out, confirmed, file=sys.stderr)
+                confirmed_out += self.specific_space + confirmed
+            print("CONFIRMED NOW:",confirmed,file=sys.stderr)
+
+
+            print(file=sys.stderr)
+            print(file=sys.stderr)
+        print("#################",file=sys.stderr)
+        if run:
+            self.buffer.insert(self.last_inserted, confirmed_out)
+            self.last_inserted = []
+
+        ret = confirmed_out
+        print("RET:",ret,file=sys.stderr)
+        return ret
+
+    def finalize(self):
+        return self.last_unconfirmed
+
+
+import argparse
+parser = argparse.ArgumentParser()
+parser.add_argument('--input-instance', type=str, default=None, help="Filename of instances to simulate input. If not set, txt input is read from stdin.")
+#parser.add_argument('--output_instance', type=str, default=None, help="Write output as instance into this file, while also writing to stdout.")
+parser.add_argument('--min-chunk-size', type=int, default=1, 
+                    help='Minimum number of space-delimited words to process in each LocalAgreement update. The more, the higher quality, but slower.')
+parser.add_argument('--min-len', type=int, default=1, 
+                    help='Minimum number of space-delimited words at the beginning.')
+#parser.add_argument('--start_at', type=int, default=0, help='Skip first N words.')
+
+# maybe later
+#parser.add_argument('--offline', action="store_true", default=False, help='Offline mode.')
+#parser.add_argument('--comp_unaware', action="store_true", default=False, help='Computationally unaware simulation.')
+
+lan_to_name = {
+    "de": "German",
+    "ja": "Japanese",
+    "zh-tr": "Chinese Traditional",
+    "zh-sim": "Chinese Simplified",
+    "cs": "Czech",
+    }
+parser.add_argument('--lan', '--language', type=str, default="de", 
+                    help="Target language code.",
+                    choices=["de", "ja","zh-tr","zh-sim","cs"])
+
+SrcLang = "English"  # always
+TgtLang = "German"
+default_prompt="You are simultaneous interpreter from {SrcLang} to {TgtLang}. We are at a conference. It is important that you translate " + \
+                "only what you hear, nothing else!"
+parser.add_argument('--sys_prompt', type=str, default=None, 
+                    help='System prompt. If None, default one is used, depending on the language. The prompt should ')
+
+default_init = "Please, go ahead, you can start with your presentation, we are ready."
+
+
+default_inits_tgt = {
+    'de': "Bitte schön, Sie können mit Ihrer Präsentation beginnen, wir sind bereit.",
+    'ja': "どうぞ、プレゼンテーションを始めてください。",  # # Please go ahead and start your presentation.  # this is in English
+    'zh-tr': "請繼續，您可以開始您的簡報，我們已經準備好了。",
+    'zh-sim': "请吧，你可以开始发言了，我们已经准备好了。",
+    'cs': "Prosím, můžete začít s prezentací, jsme připraveni.",
+}
+parser.add_argument('--init_prompt_src', type=str, default=None, help='Init translation with source text. It should be a complete sentence in the source language. ' 
+                    'It can be context specific for the given input. Default is ')
+parser.add_argument('--init_prompt_tgt', type=str, default=None, help='Init translation with this target. It should be example translation of init_prompt_src. '
+                    ' There is default init message, depending on the language.')
+
+parser.add_argument('--len-threshold', type=float, default=None, help='Ratio of the length of the source and generated target, in number of sentencepiece tokens. '
+                    'It should reflect the target language and. If not set, no len-threshold is used.')
+
+# how many times is target text longer than English
+lan_thresholds = {
+    'de': 1.3,   # 12751/9817  ... the proportion of subword tokens for ACL6060 dev de vs. en text, for EuroLLM-9B-Instruct tokenizer
+    'ja': 1.34,  # 13187/9817
+    'zh': 1.23,  # 12115/9817
+    'zh-tr': 1.23, # 12115/9817
+    'zh-sim': 1.23, # 12115/9817
+#    'cs': I don't know    # guessed
+}
+parser.add_argument('--language-specific-len-threshold', default=False, action="store_true", 
+                    help='Use language-specific length threshold, e.g. 1.3 for German.')
+
+parser.add_argument("--max-context-length", type=int, default=4096, help="Maximum number of tokens in the model to use.")
+
+parser.add_argument("--buffer_trimming", type=str, default="sentences", choices=["segments","sentences"], help="Buffer trimming strategy.")
+
+args = parser.parse_args()
+
+if args.sys_prompt is None:
+    TgtLang = lan_to_name[args.lan]
+    sys_prompt = default_prompt.format(SrcLang=SrcLang, TgtLang=TgtLang)
+else:
+    sys_prompt = args.sys_prompt
+
+if args.init_prompt_src is None:
+    init_src = default_init.split()
+    if args.init_prompt_tgt is None:
+        init_tgt = default_inits_tgt[args.lan]
+        if args.lan == "ja":
+            init_src = 'Please go ahead and start your presentation.'.split()
+            print("WARNING: Default init_prompt_src not set and language is Japanese. The init_src prompt changed to be more verbose.", file=sys.stderr)
+    else:
+        print("WARNING: init_prompt_tgt is used, init_prompt_src is None, the default one. It may be wrong!", file=sys.stderr)
+        init_tgt = args.init_prompt_tgt
+else:
+    init_src = args.init_prompt_src.split()
+    if args.init_prompt_tgt is None:
+        print("WARNING: init_prompt_src is used, init_prompt_tgt is None, so the default one is used. It may be wrong!", file=sys.stderr)
+        init_tgt = default_inits_tgt[args.lan]
+    else:
+        init_tgt = args.init_prompt_tgt
+
+print("INFO: System prompt:", sys_prompt, file=sys.stderr)
+print("INFO: Init prompt src:", init_src, file=sys.stderr)
+print("INFO: Init prompt tgt:", init_tgt, file=sys.stderr)
+
+if args.language_specific_len_threshold:
+    if args.len_threshold is not None:
+        print("ERROR: --len-threshold is set, but --language-specific-len-threshold is also set. Only one can be used.", file=sys.stderr)
+        sys.exit(1)
+    else:
+        len_threshold = lan_thresholds[args.lan]
+else:
+    len_threshold = args.len_threshold
+
+llmtrans = LLMTranslator(system_prompt=sys_prompt, max_context_length=args.max_context_length, len_ratio=len_threshold)
+lan = args.lan if not args.lan.startswith("zh") else "zh"
+simul = SimulLLM(llmtrans,language=lan, min_len=args.min_len, chunk=args.min_chunk_size,
+                init_src=init_src, init_tgt=init_tgt, trimming=args.buffer_trimming
+                )
+
+# two input options
+if args.input_instance is not None:
+    print("INFO: Reading input from file", args.input_instance, file=sys.stderr)
+    import json
+    with open(args.input_instance, "r") as f:
+        instance = json.load(f)
+
+    asr_source = instance["prediction"]
+    timestamps = instance["delays"]
+    elapsed = instance["elapsed"]
+
+    yield_ts_words = zip(timestamps, timestamps, elapsed, asr_source.split())
+else:
+    print("INFO: Reading stdin in txt format", file=sys.stderr)
+    def yield_input():
+        for line in sys.stdin:
+            line = line.strip()
+            ts, beg, end, *_ = line.split()
+            text = line[len(ts)+len(beg)+len(end)+3:]
+            ts = float(ts)
+            # in rare cases, the first word is a suffix of the previous word, that was split to multiple parts
+            if text[0] != " ":
+                first, *words = text.split()
+                yield (ts, beg, end, " "+first)  # marking the first word with " ", so that it can be later detected and inserted as suffix
+            else:
+                words = text.split()
+            for w in words:
+                yield (ts, beg, end, w)
+    yield_ts_words = yield_input()
+
+#i = 0
+for t,b,e,w in yield_ts_words:
+    if w.startswith(" "):  # it is suffix of the previous word
+        w = w[1:] 
+        simul.insert_suffix(w)
+        continue
+    simul.insert(w)
+    out = simul.process_iter()
+    if out:
+        print(t,b,e,out,flush=True)
+    # if i > 50:
+    #     break
+#    i += 1
+out = simul.finalize()
+print(t,b,e,out,flush=True)
--- a/whisperlivekit/web/live_transcription.css
+++ b/whisperlivekit/web/live_transcription.css
@@ -0,0 +1,402 @@
+:root {
+  --bg: #ffffff;
+  --text: #111111;
+  --muted: #666666;
+  --border: #e5e5e5;
+  --chip-bg: rgba(0, 0, 0, 0.04);
+  --chip-text: #000000;
+  --spinner-border: #8d8d8d5c;
+  --spinner-top: #b0b0b0;
+  --silence-bg: #f3f3f3;
+  --loading-bg: rgba(255, 77, 77, 0.06);
+  --button-bg: #ffffff;
+  --button-border: #e9e9e9;
+  --wave-stroke: #000000;
+  --label-dia-text: #868686;
+  --label-trans-text: #111111;
+}
+
+@media (prefers-color-scheme: dark) {
+  :root:not([data-theme="light"]) {
+    --bg: #0b0b0b;
+    --text: #e6e6e6;
+    --muted: #9aa0a6;
+    --border: #333333;
+    --chip-bg: rgba(255, 255, 255, 0.08);
+    --chip-text: #e6e6e6;
+    --spinner-border: #555555;
+    --spinner-top: #dddddd;
+    --silence-bg: #1a1a1a;
+    --loading-bg: rgba(255, 77, 77, 0.12);
+    --button-bg: #111111;
+    --button-border: #333333;
+    --wave-stroke: #e6e6e6;
+    --label-dia-text: #b3b3b3;
+    --label-trans-text: #ffffff;
+  }
+}
+
+:root[data-theme="dark"] {
+  --bg: #0b0b0b;
+  --text: #e6e6e6;
+  --muted: #9aa0a6;
+  --border: #333333;
+  --chip-bg: rgba(255, 255, 255, 0.08);
+  --chip-text: #e6e6e6;
+  --spinner-border: #555555;
+  --spinner-top: #dddddd;
+  --silence-bg: #1a1a1a;
+  --loading-bg: rgba(255, 77, 77, 0.12);
+  --button-bg: #111111;
+  --button-border: #333333;
+  --wave-stroke: #e6e6e6;
+  --label-dia-text: #b3b3b3;
+  --label-trans-text: #ffffff;
+}
+
+:root[data-theme="light"] {
+  --bg: #ffffff;
+  --text: #111111;
+  --muted: #666666;
+  --border: #e5e5e5;
+  --chip-bg: rgba(0, 0, 0, 0.04);
+  --chip-text: #000000;
+  --spinner-border: #8d8d8d5c;
+  --spinner-top: #b0b0b0;
+  --silence-bg: #f3f3f3;
+  --loading-bg: rgba(255, 77, 77, 0.06);
+  --button-bg: #ffffff;
+  --button-border: #e9e9e9;
+  --wave-stroke: #000000;
+  --label-dia-text: #868686;
+  --label-trans-text: #111111;
+}
+
+body {
+  font-family: ui-sans-serif, system-ui, sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', 'Noto Color Emoji';
+  margin: 20px;
+  text-align: center;
+  background-color: var(--bg);
+  color: var(--text);
+}
+
+/* Record button */
+#recordButton {
+  width: 50px;
+  height: 50px;
+  border: none;
+  border-radius: 50%;
+  background-color: var(--button-bg);
+  cursor: pointer;
+  transition: all 0.3s ease;
+  border: 1px solid var(--button-border);
+  display: flex;
+  align-items: center;
+  justify-content: center;
+  position: relative;
+}
+
+#recordButton.recording {
+  width: 180px;
+  border-radius: 40px;
+  justify-content: flex-start;
+  padding-left: 20px;
+}
+
+#recordButton:active {
+  transform: scale(0.95);
+}
+
+.shape-container {
+  width: 25px;
+  height: 25px;
+  display: flex;
+  align-items: center;
+  justify-content: center;
+  flex-shrink: 0;
+}
+
+.shape {
+  width: 25px;
+  height: 25px;
+  background-color: rgb(209, 61, 53);
+  border-radius: 50%;
+  transition: all 0.3s ease;
+}
+
+#recordButton:disabled .shape {
+  background-color: #6e6d6d;
+}
+
+#recordButton.recording .shape {
+  border-radius: 5px;
+  width: 25px;
+  height: 25px;
+}
+
+/* Recording elements */
+.recording-info {
+  display: none;
+  align-items: center;
+  margin-left: 15px;
+  flex-grow: 1;
+}
+
+#recordButton.recording .recording-info {
+  display: flex;
+}
+
+.wave-container {
+  width: 60px;
+  height: 30px;
+  position: relative;
+  display: flex;
+  align-items: center;
+  justify-content: center;
+}
+
+#waveCanvas {
+  width: 100%;
+  height: 100%;
+}
+
+.timer {
+  font-size: 14px;
+  font-weight: 500;
+  color: var(--text);
+  margin-left: 10px;
+}
+
+#status {
+  margin-top: 20px;
+  font-size: 16px;
+  color: var(--text);
+}
+
+/* Settings */
+.settings-container {
+  display: flex;
+  justify-content: center;
+  align-items: center;
+  gap: 15px;
+  margin-top: 20px;
+}
+
+.settings {
+  display: flex;
+  flex-direction: column;
+  align-items: flex-start;
+  gap: 12px;
+}
+
+.field {
+  display: flex;
+  flex-direction: column;
+  align-items: flex-start;
+  gap: 3px;
+}
+
+#chunkSelector,
+#websocketInput,
+#themeSelector {
+  font-size: 16px;
+  padding: 5px 8px;
+  border-radius: 8px;
+  border: 1px solid var(--border);
+  background-color: var(--button-bg);
+  color: var(--text);
+  max-height: 34px;
+}
+
+#websocketInput {
+  width: 220px;
+}
+
+#chunkSelector:focus,
+#websocketInput:focus,
+#themeSelector:focus {
+  outline: none;
+  border-color: #007bff;
+  box-shadow: 0 0 0 3px rgba(0, 123, 255, 0.15);
+}
+
+label {
+  font-size: 13px;
+  color: var(--muted);
+}
+
+.ws-default {
+  font-size: 12px;
+  color: var(--muted);
+}
+
+/* Segmented pill control for Theme */
+.segmented {
+  display: inline-flex;
+  align-items: stretch;
+  border: 1px solid var(--button-border);
+  background-color: var(--button-bg);
+  border-radius: 999px;
+  overflow: hidden;
+}
+
+.segmented input[type="radio"] {
+  position: absolute;
+  opacity: 0;
+  pointer-events: none;
+}
+
+.theme-selector-container {
+  position: absolute;
+  top: 20px;
+  right: 20px;
+}
+
+.segmented label {
+  display: inline-flex;
+  align-items: center;
+  gap: 6px;
+  padding: 6px 12px;
+  font-size: 14px;
+  color: var(--muted);
+  cursor: pointer;
+  user-select: none;
+  transition: background-color 0.2s ease, color 0.2s ease;
+}
+
+.segmented label span {
+  display: none;
+}
+
+.segmented label:hover span {
+  display: inline;
+}
+
+.segmented label:hover {
+  background-color: var(--chip-bg);
+}
+
+.segmented img {
+  width: 16px;
+  height: 16px;
+}
+
+.segmented input[type="radio"]:checked + label {
+  background-color: var(--chip-bg);
+  color: var(--text);
+}
+
+.segmented input[type="radio"]:focus-visible + label,
+.segmented input[type="radio"]:focus + label {
+  outline: 2px solid #007bff;
+  outline-offset: 2px;
+  border-radius: 999px;
+}
+
+/* Transcript area */
+#linesTranscript {
+  margin: 20px auto;
+  max-width: 700px;
+  text-align: left;
+  font-size: 16px;
+}
+
+#linesTranscript p {
+  margin: 0px 0;
+}
+
+#linesTranscript strong {
+  color: var(--text);
+}
+
+#speaker {
+  border: 1px solid var(--border);
+  border-radius: 100px;
+  padding: 2px 10px;
+  font-size: 14px;
+  margin-bottom: 0px;
+}
+
+.label_diarization {
+  background-color: var(--chip-bg);
+  border-radius: 8px 8px 8px 8px;
+  padding: 2px 10px;
+  margin-left: 10px;
+  display: inline-block;
+  white-space: nowrap;
+  font-size: 14px;
+  margin-bottom: 0px;
+  color: var(--label-dia-text);
+}
+
+.label_transcription {
+  background-color: var(--chip-bg);
+  border-radius: 8px 8px 8px 8px;
+  padding: 2px 10px;
+  display: inline-block;
+  white-space: nowrap;
+  margin-left: 10px;
+  font-size: 14px;
+  margin-bottom: 0px;
+  color: var(--label-trans-text);
+}
+
+#timeInfo {
+  color: var(--muted);
+  margin-left: 10px;
+}
+
+.textcontent {
+  font-size: 16px;
+  padding-left: 10px;
+  margin-bottom: 10px;
+  margin-top: 1px;
+  padding-top: 5px;
+  border-radius: 0px 0px 0px 10px;
+}
+
+.buffer_diarization {
+  color: var(--label-dia-text);
+  margin-left: 4px;
+}
+
+.buffer_transcription {
+  color: #7474748c;
+  margin-left: 4px;
+}
+
+.spinner {
+  display: inline-block;
+  width: 8px;
+  height: 8px;
+  border: 2px solid var(--spinner-border);
+  border-top: 2px solid var(--spinner-top);
+  border-radius: 50%;
+  animation: spin 0.7s linear infinite;
+  vertical-align: middle;
+  margin-bottom: 2px;
+  margin-right: 5px;
+}
+
+@keyframes spin {
+  to {
+    transform: rotate(360deg);
+  }
+}
+
+.silence {
+  color: var(--muted);
+  background-color: var(--silence-bg);
+  font-size: 13px;
+  border-radius: 30px;
+  padding: 2px 10px;
+}
+
+.loading {
+  color: var(--muted);
+  background-color: var(--loading-bg);
+  border-radius: 8px 8px 8px 0px;
+  padding: 2px 10px;
+  font-size: 14px;
+  margin-bottom: 0px;
+}
--- a/whisperlivekit/web/live_transcription.html
+++ b/whisperlivekit/web/live_transcription.html
@@ -1,861 +1,61 @@
 <!DOCTYPE html>
 <html lang="en">
-
 <head>
-    <meta charset="UTF-8" />
-    <meta name="viewport" content="width=device-width, initial-scale=1.0" />
-    <title>WhisperLiveKit</title>
-    <style>
-        :root {
-            --bg: #ffffff;
-            --text: #111111;
-            --muted: #666666;
-            --border: #e5e5e5;
-            --chip-bg: rgba(0, 0, 0, 0.04);
-            --chip-text: #000000;
-            --spinner-border: #8d8d8d5c;
-            --spinner-top: #b0b0b0;
-            --silence-bg: #f3f3f3;
-            --loading-bg: rgba(255, 77, 77, 0.06);
-            --button-bg: #ffffff;
-            --button-border: #e9e9e9;
-            --wave-stroke: #000000;
-            --label-dia-text: #868686;
-            --label-trans-text: #111111;
-        }
-
-        @media (prefers-color-scheme: dark) {
-            :root:not([data-theme="light"]) {
-                --bg: #0b0b0b;
-                --text: #e6e6e6;
-                --muted: #9aa0a6;
-                --border: #333333;
-                --chip-bg: rgba(255, 255, 255, 0.08);
-                --chip-text: #e6e6e6;
-                --spinner-border: #555555;
-                --spinner-top: #dddddd;
-                --silence-bg: #1a1a1a;
-                --loading-bg: rgba(255, 77, 77, 0.12);
-                --button-bg: #111111;
-                --button-border: #333333;
-                --wave-stroke: #e6e6e6;
-                --label-dia-text: #b3b3b3;
-                --label-trans-text: #ffffff;
-            }
-        }
-
-        :root[data-theme="dark"] {
-            --bg: #0b0b0b;
-            --text: #e6e6e6;
-            --muted: #9aa0a6;
-            --border: #333333;
-            --chip-bg: rgba(255, 255, 255, 0.08);
-            --chip-text: #e6e6e6;
-            --spinner-border: #555555;
-            --spinner-top: #dddddd;
-            --silence-bg: #1a1a1a;
-            --loading-bg: rgba(255, 77, 77, 0.12);
-            --button-bg: #111111;
-            --button-border: #333333;
-            --wave-stroke: #e6e6e6;
-            --label-dia-text: #b3b3b3;
-            --label-trans-text: #ffffff;
-        }
-
-        :root[data-theme="light"] {
-            --bg: #ffffff;
-            --text: #111111;
-            --muted: #666666;
-            --border: #e5e5e5;
-            --chip-bg: rgba(0, 0, 0, 0.04);
-            --chip-text: #000000;
-            --spinner-border: #8d8d8d5c;
-            --spinner-top: #b0b0b0;
-            --silence-bg: #f3f3f3;
-            --loading-bg: rgba(255, 77, 77, 0.06);
-            --button-bg: #ffffff;
-            --button-border: #e9e9e9;
-            --wave-stroke: #000000;
-            --label-dia-text: #868686;
-            --label-trans-text: #111111;
-        }
-        body {
-            font-family: ui-sans-serif, system-ui, sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', 'Noto Color Emoji';
-            margin: 20px;
-            text-align: center;
-            background-color: var(--bg);
-            color: var(--text);
-        }
-
-        #recordButton {
-            width: 50px;
-            height: 50px;
-            border: none;
-            border-radius: 50%;
-            background-color: var(--button-bg);
-            cursor: pointer;
-            transition: all 0.3s ease;
-            border: 1px solid var(--button-border);
-            display: flex;
-            align-items: center;
-            justify-content: center;
-            position: relative;
-        }
-
-        #recordButton.recording {
-            width: 180px;
-            border-radius: 40px;
-            justify-content: flex-start;
-            padding-left: 20px;
-        }
-
-        #recordButton:active {
-            transform: scale(0.95);
-        }
-
-        .shape-container {
-            width: 25px;
-            height: 25px;
-            display: flex;
-            align-items: center;
-            justify-content: center;
-            flex-shrink: 0;
-        }
-
-        .shape {
-            width: 25px;
-            height: 25px;
-            background-color: rgb(209, 61, 53);
-            border-radius: 50%;
-            transition: all 0.3s ease;
-        }
-
-        #recordButton:disabled .shape {
-            background-color: #6e6d6d;
-        }
-
-        #recordButton.recording .shape {
-            border-radius: 5px;
-            width: 25px;
-            height: 25px;
-        }
-
-        /* Recording elements */
-        .recording-info {
-            display: none;
-            align-items: center;
-            margin-left: 15px;
-            flex-grow: 1;
-        }
-
-        #recordButton.recording .recording-info {
-            display: flex;
-        }
-
-        .wave-container {
-            width: 60px;
-            height: 30px;
-            position: relative;
-            display: flex;
-            align-items: center;
-            justify-content: center;
-        }
-
-        #waveCanvas {
-            width: 100%;
-            height: 100%;
-        }
-
-        .timer {
-            font-size: 14px;
-            font-weight: 500;
-            color: var(--text);
-            margin-left: 10px;
-        }
-
-        #status {
-            margin-top: 20px;
-            font-size: 16px;
-            color: var(--text);
-        }
-
-        .settings-container {
-            display: flex;
-            justify-content: center;
-            align-items: center;
-            gap: 15px;
-            margin-top: 20px;
-        }
-
-        .settings {
-            display: flex;
-            flex-direction: column;
-            align-items: flex-start;
-            gap: 5px;
-        }
-
-        #chunkSelector,
-        #websocketInput,
-        #themeSelector {
-            font-size: 16px;
-            padding: 5px;
-            border-radius: 5px;
-            border: 1px solid var(--border);
-            background-color: var(--button-bg);
-            color: var(--text);
-            max-height: 30px;
-        }
-
-        #websocketInput {
-            width: 200px;
-        }
-
-        #chunkSelector:focus,
-        #websocketInput:focus,
-        #themeSelector:focus {
-            outline: none;
-            border-color: #007bff;
-        }
-
-        label {
-            font-size: 14px;
-        }
-
-        /* Speaker-labeled transcript area */
-        #linesTranscript {
-            margin: 20px auto;
-            max-width: 700px;
-            text-align: left;
-            font-size: 16px;
-        }
-
-        #linesTranscript p {
-            margin: 0px 0;
-        }
-
-        #linesTranscript strong {
-            color: var(--text);
-        }
-
-        #speaker {
-            border: 1px solid var(--border);
-            border-radius: 100px;
-            padding: 2px 10px;
-            font-size: 14px;
-            margin-bottom: 0px;
-        }
-        .label_diarization {
-            background-color: var(--chip-bg);
-            border-radius: 8px 8px 8px 8px;
-            padding: 2px 10px;
-            margin-left: 10px;
-            display: inline-block;
-            white-space: nowrap;
-            font-size: 14px;
-            margin-bottom: 0px;
-            color: var(--label-dia-text)
-        }
-
-        .label_transcription {
-            background-color: var(--chip-bg);
-            border-radius: 8px 8px 8px 8px;
-            padding: 2px 10px;
-            display: inline-block;
-            white-space: nowrap;
-            margin-left: 10px;
-            font-size: 14px;
-            margin-bottom: 0px;
-            color: var(--label-trans-text)
-        }
-
-        #timeInfo {
-            color: var(--muted);
-            margin-left: 10px;
-        }
-
-        .textcontent {
-            font-size: 16px;
-            /* margin-left: 10px; */
-            padding-left: 10px;
-            margin-bottom: 10px;
-            margin-top: 1px;
-            padding-top: 5px;
-            border-radius: 0px 0px 0px 10px;
-        }
-
-        .buffer_diarization {
-            color: var(--label-dia-text);
-            margin-left: 4px;
-        }
-
-        .buffer_transcription {
-            color: #7474748c;
-            margin-left: 4px;
-        }
-
-
-        .spinner {
-            display: inline-block;
-            width: 8px;
-            height: 8px;
-            border: 2px solid var(--spinner-border);
-            border-top: 2px solid var(--spinner-top);
-            border-radius: 50%;
-            animation: spin 0.7s linear infinite;
-            vertical-align: middle;
-            margin-bottom: 2px;
-            margin-right: 5px;
-        }
-
-        @keyframes spin {
-            to {
-                transform: rotate(360deg);
-            }
-        }
-
-        .silence {
-            color: var(--muted);
-            background-color: var(--silence-bg);
-            font-size: 13px;
-            border-radius: 30px;
-            padding: 2px 10px;
-        }
-
-        .loading {
-            color: var(--muted);
-            background-color: var(--loading-bg);
-            border-radius: 8px 8px 8px 0px;
-            padding: 2px 10px;
-            font-size: 14px;
-            margin-bottom: 0px;
-        }
-    </style>
+  <meta charset="UTF-8" />
+  <meta name="viewport" content="width=device-width, initial-scale=1.0" />
+  <title>WhisperLiveKit</title>
+  <link rel="stylesheet" href="/web/live_transcription.css" />
 </head>
-
 <body>
-
-    <div class="settings-container">
-        <button id="recordButton">
-            <div class="shape-container">
-                <div class="shape"></div>
-            </div>
-            <div class="recording-info">
-                <div class="wave-container">
-                    <canvas id="waveCanvas"></canvas>
-                </div>
-                <div class="timer">00:00</div>
-            </div>
-        </button>
-        <div class="settings">
-            <div>
-                <label for="chunkSelector">Chunk size (ms):</label>
-                <select id="chunkSelector">
-                    <option value="500">500 ms</option>
-                    <option value="1000" selected>1000 ms</option>
-                    <option value="2000">2000 ms</option>
-                    <option value="3000">3000 ms</option>
-                    <option value="4000">4000 ms</option>
-                    <option value="5000">5000 ms</option>
-                </select>
-            </div>
-            <div>
-                <label for="websocketInput">WebSocket URL:</label>
-                <input id="websocketInput" type="text" />
-            </div>
-            <div>
-                <label for="themeSelector">Theme:</label>
-                <select id="themeSelector">
-                    <option value="system" selected>System</option>
-                    <option value="light">Light</option>
-                    <option value="dark">Dark</option>
-                </select>
-            </div>
+  <div class="settings-container">
+    <button id="recordButton">
+      <div class="shape-container">
+        <div class="shape"></div>
+      </div>
+      <div class="recording-info">
+        <div class="wave-container">
+          <canvas id="waveCanvas"></canvas>
        </div>
+        <div class="timer">00:00</div>
+      </div>
+    </button>
+
+    <div class="settings">
+      <div class="field">
+        <label for="websocketInput">WebSocket URL</label>
+        <input id="websocketInput" type="text" placeholder="ws://host:port/asr" />
+      </div>
+
+      </div>
    </div>
+  </div>

-    <p id="status"></p>
+  <div class="theme-selector-container">
+    <div class="segmented" role="radiogroup" aria-label="Theme selector">
+      <input type="radio" id="theme-system" name="theme" value="system" />
+      <label for="theme-system" title="System">
+        <img src="/web/src/system_mode.svg" alt="" />
+        <span>System</span>
+      </label>

-    <!-- Speaker-labeled transcript -->
-    <div id="linesTranscript"></div>
+      <input type="radio" id="theme-light" name="theme" value="light" />
+      <label for="theme-light" title="Light">
+        <img src="/web/src/light_mode.svg" alt="" />
+        <span>Light</span>
+      </label>

-    <script>
-        let isRecording = false;
-        let websocket = null;
-        let recorder = null;
-        let chunkDuration = 1000;
-        let websocketUrl = "ws://localhost:8000/asr";
-        let userClosing = false;
-        let wakeLock = null;
-        let startTime = null;
-        let timerInterval = null;
-        let audioContext = null;
-        let analyser = null;
-        let microphone = null;
-        let waveCanvas = document.getElementById("waveCanvas");
-        let waveCtx = waveCanvas.getContext("2d");
-        let animationFrame = null;
-        let waitingForStop = false;
-        let lastReceivedData = null;
-        let lastSignature = null;
-        waveCanvas.width = 60 * (window.devicePixelRatio || 1);
-        waveCanvas.height = 30 * (window.devicePixelRatio || 1);
-        waveCtx.scale(window.devicePixelRatio || 1, window.devicePixelRatio || 1);
+      <input type="radio" id="theme-dark" name="theme" value="dark" />
+      <label for="theme-dark" title="Dark">
+        <img src="/web/src/dark_mode.svg" alt="" />
+        <span>Dark</span>
+      </label>
+    </div>
+  </div>

-        const statusText = document.getElementById("status");
-        const recordButton = document.getElementById("recordButton");
-        const chunkSelector = document.getElementById("chunkSelector");
-        const websocketInput = document.getElementById("websocketInput");
-        const linesTranscriptDiv = document.getElementById("linesTranscript");
-        const timerElement = document.querySelector(".timer");
-        const themeSelector = document.getElementById("themeSelector");
+  <p id="status"></p>

-        function getWaveStroke() {
-            const styles = getComputedStyle(document.documentElement);
-            const v = styles.getPropertyValue("--wave-stroke").trim();
-            return v || "#000";
-        }
+  <div id="linesTranscript"></div>

-        let waveStroke = getWaveStroke();
-
-        function updateWaveStroke() {
-            waveStroke = getWaveStroke();
-        }
-
-        function applyTheme(pref) {
-            if (pref === "light") {
-                document.documentElement.setAttribute("data-theme", "light");
-            } else if (pref === "dark") {
-                document.documentElement.setAttribute("data-theme", "dark");
-            } else {
-                document.documentElement.removeAttribute("data-theme");
-            }
-            updateWaveStroke();
-        }
-
-        const savedThemePref = localStorage.getItem("themePreference") || "system";
-        applyTheme(savedThemePref);
-        if (themeSelector) {
-            themeSelector.value = savedThemePref;
-            themeSelector.addEventListener("change", () => {
-                const val = themeSelector.value;
-                localStorage.setItem("themePreference", val);
-                applyTheme(val);
-            });
-        }
-
-        const darkMq = window.matchMedia && window.matchMedia("(prefers-color-scheme: dark)");
-        const handleOsThemeChange = () => {
-            const pref = localStorage.getItem("themePreference") || "system";
-            if (pref === "system") updateWaveStroke();
-        };
-        if (darkMq && darkMq.addEventListener) {
-            darkMq.addEventListener("change", handleOsThemeChange);
-        } else if (darkMq && darkMq.addListener) {
-            darkMq.addListener(handleOsThemeChange);
-        }
-
-        function fmt1(x) {
-            const n = Number(x);
-            return Number.isFinite(n) ? n.toFixed(1) : x;
-        }
-
-        const host = window.location.hostname || "localhost";
-        const port = window.location.port;
-        const protocol = window.location.protocol === "https:" ? "wss" : "ws";
-        const defaultWebSocketUrl = `${protocol}://${host}:${port}/asr`;
-        websocketInput.value = defaultWebSocketUrl;
-        websocketUrl = defaultWebSocketUrl;
-
-        chunkSelector.addEventListener("change", () => {
-            chunkDuration = parseInt(chunkSelector.value);
-        });
-
-        websocketInput.addEventListener("change", () => {
-            const urlValue = websocketInput.value.trim();
-            if (!urlValue.startsWith("ws://") && !urlValue.startsWith("wss://")) {
-                statusText.textContent = "Invalid WebSocket URL (must start with ws:// or wss://)";
-                return;
-            }
-            websocketUrl = urlValue;
-            statusText.textContent = "WebSocket URL updated. Ready to connect.";
-        });
-
-        function setupWebSocket() {
-            return new Promise((resolve, reject) => {
-                try {
-                    websocket = new WebSocket(websocketUrl);
-                } catch (error) {
-                    statusText.textContent = "Invalid WebSocket URL. Please check and try again.";
-                    reject(error);
-                    return;
-                }
-
-                websocket.onopen = () => {
-                    statusText.textContent = "Connected to server.";
-                    resolve();
-                };
-
-                websocket.onclose = () => {
-                    if (userClosing) {
-                        if (waitingForStop) {
-                            statusText.textContent = "Processing finalized or connection closed.";
-                            if (lastReceivedData) {
-                                renderLinesWithBuffer(
-                                    lastReceivedData.lines || [],
-                                    lastReceivedData.buffer_diarization || "",
-                                    lastReceivedData.buffer_transcription || "",
-                                    0, 0, true // isFinalizing = true
-                                );
-                            }
-                        }
-                        // If ready_to_stop was received, statusText is already "Finished processing..."
-                        // and waitingForStop is false.
-                    } else {
-                        statusText.textContent = "Disconnected from the WebSocket server. (Check logs if model is loading.)";
-                        if (isRecording) {
-                            stopRecording(); 
-                        }
-                    }
-                    isRecording = false;  
-                    waitingForStop = false; 
-                    userClosing = false;  
-                    lastReceivedData = null;  
-                    websocket = null;    
-                    updateUI();  
-                };
-
-                websocket.onerror = () => {
-                    statusText.textContent = "Error connecting to WebSocket.";
-                    reject(new Error("Error connecting to WebSocket"));
-                };
-
-                // Handle messages from server
-                websocket.onmessage = (event) => {
-                    const data = JSON.parse(event.data);
-                    
-                    // Check for status messages
-                    if (data.type === "ready_to_stop") {
-                        console.log("Ready to stop received, finalizing display and closing WebSocket.");
-                        waitingForStop = false;
-
-                        if (lastReceivedData) {
-                            renderLinesWithBuffer(
-                                lastReceivedData.lines || [],
-                                lastReceivedData.buffer_diarization || "",
-                                lastReceivedData.buffer_transcription || "",
-                                0, // No more lag
-                                0, // No more lag
-                                true // isFinalizing = true
-                            );
-                        }
-                        statusText.textContent = "Finished processing audio! Ready to record again.";
-                        recordButton.disabled = false;
-                        
-                        if (websocket) {
-                            websocket.close(); // will trigger onclose
-                            // websocket = null; // onclose handle setting websocket to null
-                        }
-                        return;
-                    }
-                    
-                    lastReceivedData = data; 
-                    
-                    // Handle normal transcription updates
-                    const { 
-                        lines = [], 
-                        buffer_transcription = "", 
-                        buffer_diarization = "",
-                        remaining_time_transcription = 0,
-                        remaining_time_diarization = 0,
-                        status = "active_transcription"
-                    } = data;
-                    
-                    renderLinesWithBuffer(
-                        lines, 
-                        buffer_diarization, 
-                        buffer_transcription, 
-                        remaining_time_diarization,
-                        remaining_time_transcription,
-                        false,
-                        status
-                    );
-                };
-            });
-        }
-
-        function renderLinesWithBuffer(lines, buffer_diarization, buffer_transcription, remaining_time_diarization, remaining_time_transcription, isFinalizing = false, current_status = "active_transcription") {
-            if (current_status === "no_audio_detected") {
-                linesTranscriptDiv.innerHTML = "<p style='text-align: center; color: var(--muted); margin-top: 20px;'><em>No audio detected...</em></p>";
-                return; 
-            }
-
-            // try to keep stable DOM despite having updates every 0.1s. only update numeric lag values if structure hasn't changed
-            const showLoading = (!isFinalizing) && (lines || []).some(it => it.speaker == 0);
-            const showTransLag = !isFinalizing && remaining_time_transcription > 0;
-            const showDiaLag = !isFinalizing && !!buffer_diarization && remaining_time_diarization > 0;
-            const signature = JSON.stringify({
-                lines: (lines || []).map(it => ({ speaker: it.speaker, text: it.text, beg: it.beg, end: it.end })),
-                buffer_transcription: buffer_transcription || "",
-                buffer_diarization: buffer_diarization || "",
-                status: current_status,
-                showLoading,
-                showTransLag,
-                showDiaLag,
-                isFinalizing: !!isFinalizing
-            });
-            if (lastSignature === signature) {
-                const t = document.querySelector(".lag-transcription-value");
-                if (t) t.textContent = fmt1(remaining_time_transcription);
-                const d = document.querySelector(".lag-diarization-value");
-                if (d) d.textContent = fmt1(remaining_time_diarization);
-                const ld = document.querySelector(".loading-diarization-value");
-                if (ld) ld.textContent = fmt1(remaining_time_diarization);
-                return;
-            }
-            lastSignature = signature;
-
-            const linesHtml = lines.map((item, idx) => {
-                let timeInfo = "";
-                if (item.beg !== undefined && item.end !== undefined) {
-                    timeInfo = ` ${item.beg} - ${item.end}`;
-                }
-
-                let speakerLabel = "";
-                if (item.speaker === -2) {
-                    speakerLabel = `<span class="silence">Silence<span id='timeInfo'>${timeInfo}</span></span>`;
-                } else if (item.speaker == 0 && !isFinalizing) {
-                    speakerLabel = `<span class='loading'><span class="spinner"></span><span id='timeInfo'><span class="loading-diarization-value">${fmt1(remaining_time_diarization)}</span> second(s) of audio are undergoing diarization</span></span>`;
-                } else if (item.speaker == -1) {
-                    speakerLabel = `<span id="speaker">Speaker 1<span id='timeInfo'>${timeInfo}</span></span>`;
-                } else if (item.speaker !== -1 && item.speaker !== 0) {
-                    speakerLabel = `<span id="speaker">Speaker ${item.speaker}<span id='timeInfo'>${timeInfo}</span></span>`;
-                }
-
-
-                let currentLineText = item.text || "";
-
-                if (idx === lines.length - 1) { 
-                    if (!isFinalizing && item.speaker !== -2) {
-                        if (remaining_time_transcription > 0) {
-                             speakerLabel += `<span class="label_transcription"><span class="spinner"></span>Transcription lag <span id='timeInfo'><span class="lag-transcription-value">${fmt1(remaining_time_transcription)}</span>s</span></span>`;
-                        }
-                        if (buffer_diarization && remaining_time_diarization > 0) {
-                             speakerLabel += `<span class="label_diarization"><span class="spinner"></span>Diarization lag<span id='timeInfo'><span class="lag-diarization-value">${fmt1(remaining_time_diarization)}</span>s</span></span>`;
-                        }
-                    }
-
-                    if (buffer_diarization) {
-                        if (isFinalizing) {
-                            currentLineText += (currentLineText.length > 0 && buffer_diarization.trim().length > 0 ? " " : "") + buffer_diarization.trim();
-                        } else {
-                            currentLineText += `<span class="buffer_diarization">${buffer_diarization}</span>`;
-                        }
-                    }
-                    if (buffer_transcription) {
-                        if (isFinalizing) {
-                            currentLineText += (currentLineText.length > 0 && buffer_transcription.trim().length > 0 ? " " : "") + buffer_transcription.trim();
-                        } else {
-                            currentLineText += `<span class="buffer_transcription">${buffer_transcription}</span>`;
-                        }
-                    }
-                }
-                
-                return currentLineText.trim().length > 0 || speakerLabel.length > 0
-                    ? `<p>${speakerLabel}<br/><div class='textcontent'>${currentLineText}</div></p>`
-                    : `<p>${speakerLabel}<br/></p>`; 
-            }).join("");
-
-            linesTranscriptDiv.innerHTML = linesHtml;
-            window.scrollTo({ top: document.body.scrollHeight, behavior: 'smooth' });
-        }
-
-        function updateTimer() {
-            if (!startTime) return;
-            
-            const elapsed = Math.floor((Date.now() - startTime) / 1000);
-            const minutes = Math.floor(elapsed / 60).toString().padStart(2, "0");
-            const seconds = (elapsed % 60).toString().padStart(2, "0");
-            timerElement.textContent = `${minutes}:${seconds}`;
-        }
-
-        function drawWaveform() {
-            if (!analyser) return;
-            
-            const bufferLength = analyser.frequencyBinCount;
-            const dataArray = new Uint8Array(bufferLength);
-            analyser.getByteTimeDomainData(dataArray);
-            
-            waveCtx.clearRect(0, 0, waveCanvas.width / (window.devicePixelRatio || 1), waveCanvas.height / (window.devicePixelRatio || 1));
-            waveCtx.lineWidth = 1;
-            waveCtx.strokeStyle = waveStroke;
-            waveCtx.beginPath();
-            
-            const sliceWidth = (waveCanvas.width / (window.devicePixelRatio || 1)) / bufferLength;
-            let x = 0;
-            
-            for (let i = 0; i < bufferLength; i++) {
-                const v = dataArray[i] / 128.0;
-                const y = v * (waveCanvas.height / (window.devicePixelRatio || 1)) / 2;
-                
-                if (i === 0) {
-                    waveCtx.moveTo(x, y);
-                } else {
-                    waveCtx.lineTo(x, y);
-                }
-                
-                x += sliceWidth;
-            }
-            
-            waveCtx.lineTo(waveCanvas.width / (window.devicePixelRatio || 1), waveCanvas.height / (window.devicePixelRatio || 1) / 2);
-            waveCtx.stroke();
-            
-            animationFrame = requestAnimationFrame(drawWaveform);
-        }
-
-        async function startRecording() {
-            try {
-
-                // https://developer.mozilla.org/en-US/docs/Web/API/Screen_Wake_Lock_API
-                // create an async function to request a wake lock
-                try {
-                  wakeLock = await navigator.wakeLock.request("screen");
-                } catch (err) {
-                  // The Wake Lock request has failed - usually system related, such as battery.
-                  console.log("Error acquiring wake lock.")
-                }
-
-                const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
-                
-                audioContext = new (window.AudioContext || window.webkitAudioContext)();
-                analyser = audioContext.createAnalyser();
-                analyser.fftSize = 256;
-                microphone = audioContext.createMediaStreamSource(stream);
-                microphone.connect(analyser);
-                
-                recorder = new MediaRecorder(stream, { mimeType: "audio/webm" });
-                recorder.ondataavailable = (e) => {
-                    if (websocket && websocket.readyState === WebSocket.OPEN) {
-                        websocket.send(e.data);
-                    }
-                };
-                recorder.start(chunkDuration);
-                
-                startTime = Date.now();
-                timerInterval = setInterval(updateTimer, 1000);
-                drawWaveform();
-                
-                isRecording = true;
-                updateUI();
-            } catch (err) {
-                statusText.textContent = "Error accessing microphone. Please allow microphone access.";
-                console.error(err);
-            }
-        }
-
-        async function stopRecording() {
-            wakeLock.release().then(() => {
-              wakeLock = null;
-            });
-  
-            userClosing = true;
-            waitingForStop = true;
-            
-            if (websocket && websocket.readyState === WebSocket.OPEN) {
-                // Send empty audio buffer as stop signal
-                const emptyBlob = new Blob([], { type: 'audio/webm' });
-                websocket.send(emptyBlob);
-                statusText.textContent = "Recording stopped. Processing final audio...";
-            }
-            
-            if (recorder) {
-                recorder.stop();
-                recorder = null;
-            }
-            
-            if (microphone) {
-                microphone.disconnect();
-                microphone = null;
-            }
-            
-            if (analyser) {
-                analyser = null;
-            }
-            
-            if (audioContext && audioContext.state !== 'closed') {
-                try {
-                    audioContext.close();
-                } catch (e) {
-                    console.warn("Could not close audio context:", e);
-                }
-                audioContext = null;
-            }
-            
-            if (animationFrame) {
-                cancelAnimationFrame(animationFrame);
-                animationFrame = null;
-            }
-            
-            if (timerInterval) {
-                clearInterval(timerInterval);
-                timerInterval = null;
-            }            
-            timerElement.textContent = "00:00";
-            startTime = null;
-            
-            
-            isRecording = false;
-            updateUI();	
-        }
-
-        async function toggleRecording() {
-            if (!isRecording) {
-                if (waitingForStop) {
-                    console.log("Waiting for stop, early return");
-                    return;  // Early return, UI is already updated
-                }
-                console.log("Connecting to WebSocket");
-                try {
-                    // If we have an active WebSocket that's still processing, just restart audio capture
-                    if (websocket && websocket.readyState === WebSocket.OPEN) {
-                        await startRecording();
-                    } else {
-                        // If no active WebSocket or it's closed, create new one
-                        await setupWebSocket();
-                        await startRecording();
-                    }
-                } catch (err) {
-                    statusText.textContent = "Could not connect to WebSocket or access mic. Aborted.";
-                    console.error(err);
-                }
-            } else {
-                console.log("Stopping recording");
-                stopRecording();
-            }
-        }
-
-        function updateUI() {
-            recordButton.classList.toggle("recording", isRecording);
-            recordButton.disabled = waitingForStop;
-
-            if (waitingForStop) {
-                if (statusText.textContent !== "Recording stopped. Processing final audio...") {
-                     statusText.textContent = "Please wait for processing to complete...";
-                }
-            } else if (isRecording) {
-                statusText.textContent = "Recording...";
-            } else {
-                if (statusText.textContent !== "Finished processing audio! Ready to record again." &&
-                    statusText.textContent !== "Processing finalized or connection closed.") {
-                    statusText.textContent = "Click to start transcription";
-                }
-            }
-            if (!waitingForStop) {
-                recordButton.disabled = false;
-            }
-        }
-
-        recordButton.addEventListener("click", toggleRecording);
-    </script>
+  <script src="/web/live_transcription.js"></script>
 </body>
-
 </html>
--- a/whisperlivekit/web/live_transcription.js
+++ b/whisperlivekit/web/live_transcription.js
@@ -0,0 +1,518 @@
+/* Theme, WebSocket, recording, rendering logic extracted from inline script and adapted for segmented theme control and WS caption */
+
+let isRecording = false;
+let websocket = null;
+let recorder = null;
+let chunkDuration = 100;
+let websocketUrl = "ws://localhost:8000/asr";
+let userClosing = false;
+let wakeLock = null;
+let startTime = null;
+let timerInterval = null;
+let audioContext = null;
+let analyser = null;
+let microphone = null;
+let waveCanvas = document.getElementById("waveCanvas");
+let waveCtx = waveCanvas.getContext("2d");
+let animationFrame = null;
+let waitingForStop = false;
+let lastReceivedData = null;
+let lastSignature = null;
+
+waveCanvas.width = 60 * (window.devicePixelRatio || 1);
+waveCanvas.height = 30 * (window.devicePixelRatio || 1);
+waveCtx.scale(window.devicePixelRatio || 1, window.devicePixelRatio || 1);
+
+const statusText = document.getElementById("status");
+const recordButton = document.getElementById("recordButton");
+const chunkSelector = document.getElementById("chunkSelector");
+const websocketInput = document.getElementById("websocketInput");
+const websocketDefaultSpan = document.getElementById("wsDefaultUrl");
+const linesTranscriptDiv = document.getElementById("linesTranscript");
+const timerElement = document.querySelector(".timer");
+const themeRadios = document.querySelectorAll('input[name="theme"]');
+
+function getWaveStroke() {
+  const styles = getComputedStyle(document.documentElement);
+  const v = styles.getPropertyValue("--wave-stroke").trim();
+  return v || "#000";
+}
+
+let waveStroke = getWaveStroke();
+function updateWaveStroke() {
+  waveStroke = getWaveStroke();
+}
+
+function applyTheme(pref) {
+  if (pref === "light") {
+    document.documentElement.setAttribute("data-theme", "light");
+  } else if (pref === "dark") {
+    document.documentElement.setAttribute("data-theme", "dark");
+  } else {
+    document.documentElement.removeAttribute("data-theme");
+  }
+  updateWaveStroke();
+}
+
+// Persisted theme preference
+const savedThemePref = localStorage.getItem("themePreference") || "system";
+applyTheme(savedThemePref);
+if (themeRadios.length) {
+  themeRadios.forEach((r) => {
+    r.checked = r.value === savedThemePref;
+    r.addEventListener("change", () => {
+      if (r.checked) {
+        localStorage.setItem("themePreference", r.value);
+        applyTheme(r.value);
+      }
+    });
+  });
+}
+
+// React to OS theme changes when in "system" mode
+const darkMq = window.matchMedia && window.matchMedia("(prefers-color-scheme: dark)");
+const handleOsThemeChange = () => {
+  const pref = localStorage.getItem("themePreference") || "system";
+  if (pref === "system") updateWaveStroke();
+};
+if (darkMq && darkMq.addEventListener) {
+  darkMq.addEventListener("change", handleOsThemeChange);
+} else if (darkMq && darkMq.addListener) {
+  // deprecated, but included for Safari compatibility
+  darkMq.addListener(handleOsThemeChange);
+}
+
+// Helpers
+function fmt1(x) {
+  const n = Number(x);
+  return Number.isFinite(n) ? n.toFixed(1) : x;
+}
+
+// Default WebSocket URL computation
+const host = window.location.hostname || "localhost";
+const port = window.location.port;
+const protocol = window.location.protocol === "https:" ? "wss" : "ws";
+const defaultWebSocketUrl = `${protocol}://${host}${port ? ":" + port : ""}/asr`;
+
+// Populate default caption and input
+if (websocketDefaultSpan) websocketDefaultSpan.textContent = defaultWebSocketUrl;
+websocketInput.value = defaultWebSocketUrl;
+websocketUrl = defaultWebSocketUrl;
+
+// Optional chunk selector (guard for presence)
+if (chunkSelector) {
+  chunkSelector.addEventListener("change", () => {
+    chunkDuration = parseInt(chunkSelector.value);
+  });
+}
+
+// WebSocket input change handling
+websocketInput.addEventListener("change", () => {
+  const urlValue = websocketInput.value.trim();
+  if (!urlValue.startsWith("ws://") && !urlValue.startsWith("wss://")) {
+    statusText.textContent = "Invalid WebSocket URL (must start with ws:// or wss://)";
+    return;
+  }
+  websocketUrl = urlValue;
+  statusText.textContent = "WebSocket URL updated. Ready to connect.";
+});
+
+function setupWebSocket() {
+  return new Promise((resolve, reject) => {
+    try {
+      websocket = new WebSocket(websocketUrl);
+    } catch (error) {
+      statusText.textContent = "Invalid WebSocket URL. Please check and try again.";
+      reject(error);
+      return;
+    }
+
+    websocket.onopen = () => {
+      statusText.textContent = "Connected to server.";
+      resolve();
+    };
+
+    websocket.onclose = () => {
+      if (userClosing) {
+        if (waitingForStop) {
+          statusText.textContent = "Processing finalized or connection closed.";
+          if (lastReceivedData) {
+            renderLinesWithBuffer(
+              lastReceivedData.lines || [],
+              lastReceivedData.buffer_diarization || "",
+              lastReceivedData.buffer_transcription || "",
+              0,
+              0,
+              true
+            );
+          }
+        }
+      } else {
+        statusText.textContent = "Disconnected from the WebSocket server. (Check logs if model is loading.)";
+        if (isRecording) {
+          stopRecording();
+        }
+      }
+      isRecording = false;
+      waitingForStop = false;
+      userClosing = false;
+      lastReceivedData = null;
+      websocket = null;
+      updateUI();
+    };
+
+    websocket.onerror = () => {
+      statusText.textContent = "Error connecting to WebSocket.";
+      reject(new Error("Error connecting to WebSocket"));
+    };
+
+    websocket.onmessage = (event) => {
+      const data = JSON.parse(event.data);
+
+      if (data.type === "ready_to_stop") {
+        console.log("Ready to stop received, finalizing display and closing WebSocket.");
+        waitingForStop = false;
+
+        if (lastReceivedData) {
+          renderLinesWithBuffer(
+            lastReceivedData.lines || [],
+            lastReceivedData.buffer_diarization || "",
+            lastReceivedData.buffer_transcription || "",
+            0,
+            0,
+            true
+          );
+        }
+        statusText.textContent = "Finished processing audio! Ready to record again.";
+        recordButton.disabled = false;
+
+        if (websocket) {
+          websocket.close();
+        }
+        return;
+      }
+
+      lastReceivedData = data;
+
+      const {
+        lines = [],
+        buffer_transcription = "",
+        buffer_diarization = "",
+        remaining_time_transcription = 0,
+        remaining_time_diarization = 0,
+        status = "active_transcription",
+      } = data;
+
+      renderLinesWithBuffer(
+        lines,
+        buffer_diarization,
+        buffer_transcription,
+        remaining_time_diarization,
+        remaining_time_transcription,
+        false,
+        status
+      );
+    };
+  });
+}
+
+function renderLinesWithBuffer(
+  lines,
+  buffer_diarization,
+  buffer_transcription,
+  remaining_time_diarization,
+  remaining_time_transcription,
+  isFinalizing = false,
+  current_status = "active_transcription"
+) {
+  if (current_status === "no_audio_detected") {
+    linesTranscriptDiv.innerHTML =
+      "<p style='text-align: center; color: var(--muted); margin-top: 20px;'><em>No audio detected...</em></p>";
+    return;
+  }
+
+  const showLoading = !isFinalizing && (lines || []).some((it) => it.speaker == 0);
+  const showTransLag = !isFinalizing && remaining_time_transcription > 0;
+  const showDiaLag = !isFinalizing && !!buffer_diarization && remaining_time_diarization > 0;
+  const signature = JSON.stringify({
+    lines: (lines || []).map((it) => ({ speaker: it.speaker, text: it.text, beg: it.beg, end: it.end })),
+    buffer_transcription: buffer_transcription || "",
+    buffer_diarization: buffer_diarization || "",
+    status: current_status,
+    showLoading,
+    showTransLag,
+    showDiaLag,
+    isFinalizing: !!isFinalizing,
+  });
+  if (lastSignature === signature) {
+    const t = document.querySelector(".lag-transcription-value");
+    if (t) t.textContent = fmt1(remaining_time_transcription);
+    const d = document.querySelector(".lag-diarization-value");
+    if (d) d.textContent = fmt1(remaining_time_diarization);
+    const ld = document.querySelector(".loading-diarization-value");
+    if (ld) ld.textContent = fmt1(remaining_time_diarization);
+    return;
+  }
+  lastSignature = signature;
+
+  const linesHtml = (lines || [])
+    .map((item, idx) => {
+      let timeInfo = "";
+      if (item.beg !== undefined && item.end !== undefined) {
+        timeInfo = ` ${item.beg} - ${item.end}`;
+      }
+
+      let speakerLabel = "";
+      if (item.speaker === -2) {
+        speakerLabel = `<span class="silence">Silence<span id='timeInfo'>${timeInfo}</span></span>`;
+      } else if (item.speaker == 0 && !isFinalizing) {
+        speakerLabel = `<span class='loading'><span class="spinner"></span><span id='timeInfo'><span class="loading-diarization-value">${fmt1(
+          remaining_time_diarization
+        )}</span> second(s) of audio are undergoing diarization</span></span>`;
+      } else if (item.speaker !== 0) {
+        speakerLabel = `<span id="speaker">Speaker ${item.speaker}<span id='timeInfo'>${timeInfo}</span></span>`;
+      }
+
+      let currentLineText = item.text || "";
+
+      if (idx === lines.length - 1) {
+        if (!isFinalizing && item.speaker !== -2) {
+          if (remaining_time_transcription > 0) {
+            speakerLabel += `<span class="label_transcription"><span class="spinner"></span>Transcription lag <span id='timeInfo'><span class="lag-transcription-value">${fmt1(
+              remaining_time_transcription
+            )}</span>s</span></span>`;
+          }
+          if (buffer_diarization && remaining_time_diarization > 0) {
+            speakerLabel += `<span class="label_diarization"><span class="spinner"></span>Diarization lag<span id='timeInfo'><span class="lag-diarization-value">${fmt1(
+              remaining_time_diarization
+            )}</span>s</span></span>`;
+          }
+        }
+
+        if (buffer_diarization) {
+          if (isFinalizing) {
+            currentLineText +=
+              (currentLineText.length > 0 && buffer_diarization.trim().length > 0 ? " " : "") + buffer_diarization.trim();
+          } else {
+            currentLineText += `<span class="buffer_diarization">${buffer_diarization}</span>`;
+          }
+        }
+        if (buffer_transcription) {
+          if (isFinalizing) {
+            currentLineText +=
+              (currentLineText.length > 0 && buffer_transcription.trim().length > 0 ? " " : "") +
+              buffer_transcription.trim();
+          } else {
+            currentLineText += `<span class="buffer_transcription">${buffer_transcription}</span>`;
+          }
+        }
+      }
+
+      return currentLineText.trim().length > 0 || speakerLabel.length > 0
+        ? `<p>${speakerLabel}<br/><div class='textcontent'>${currentLineText}</div></p>`
+        : `<p>${speakerLabel}<br/></p>`;
+    })
+    .join("");
+
+  linesTranscriptDiv.innerHTML = linesHtml;
+  window.scrollTo({ top: document.body.scrollHeight, behavior: "smooth" });
+}
+
+function updateTimer() {
+  if (!startTime) return;
+
+  const elapsed = Math.floor((Date.now() - startTime) / 1000);
+  const minutes = Math.floor(elapsed / 60).toString().padStart(2, "0");
+  const seconds = (elapsed % 60).toString().padStart(2, "0");
+  timerElement.textContent = `${minutes}:${seconds}`;
+}
+
+function drawWaveform() {
+  if (!analyser) return;
+
+  const bufferLength = analyser.frequencyBinCount;
+  const dataArray = new Uint8Array(bufferLength);
+  analyser.getByteTimeDomainData(dataArray);
+
+  waveCtx.clearRect(
+    0,
+    0,
+    waveCanvas.width / (window.devicePixelRatio || 1),
+    waveCanvas.height / (window.devicePixelRatio || 1)
+  );
+  waveCtx.lineWidth = 1;
+  waveCtx.strokeStyle = waveStroke;
+  waveCtx.beginPath();
+
+  const sliceWidth = (waveCanvas.width / (window.devicePixelRatio || 1)) / bufferLength;
+  let x = 0;
+
+  for (let i = 0; i < bufferLength; i++) {
+    const v = dataArray[i] / 128.0;
+    const y = (v * (waveCanvas.height / (window.devicePixelRatio || 1))) / 2;
+
+    if (i === 0) {
+      waveCtx.moveTo(x, y);
+    } else {
+      waveCtx.lineTo(x, y);
+    }
+
+    x += sliceWidth;
+  }
+
+  waveCtx.lineTo(
+    waveCanvas.width / (window.devicePixelRatio || 1),
+    (waveCanvas.height / (window.devicePixelRatio || 1)) / 2
+  );
+  waveCtx.stroke();
+
+  animationFrame = requestAnimationFrame(drawWaveform);
+}
+
+async function startRecording() {
+  try {
+    try {
+      wakeLock = await navigator.wakeLock.request("screen");
+    } catch (err) {
+      console.log("Error acquiring wake lock.");
+    }
+
+    const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
+
+    audioContext = new (window.AudioContext || window.webkitAudioContext)();
+    analyser = audioContext.createAnalyser();
+    analyser.fftSize = 256;
+    microphone = audioContext.createMediaStreamSource(stream);
+    microphone.connect(analyser);
+
+    recorder = new MediaRecorder(stream, { mimeType: "audio/webm" });
+    recorder.ondataavailable = (e) => {
+      if (websocket && websocket.readyState === WebSocket.OPEN) {
+        websocket.send(e.data);
+      }
+    };
+    recorder.start(chunkDuration);
+
+    startTime = Date.now();
+    timerInterval = setInterval(updateTimer, 1000);
+    drawWaveform();
+
+    isRecording = true;
+    updateUI();
+  } catch (err) {
+    if (window.location.hostname === "0.0.0.0") {
+      statusText.textContent =
+        "Error accessing microphone. Browsers may block microphone access on 0.0.0.0. Try using localhost:8000 instead.";
+    } else {
+      statusText.textContent = "Error accessing microphone. Please allow microphone access.";
+    }
+    console.error(err);
+  }
+}
+
+async function stopRecording() {
+  if (wakeLock) {
+    try {
+      await wakeLock.release();
+    } catch (e) {
+      // ignore
+    }
+    wakeLock = null;
+  }
+
+  userClosing = true;
+  waitingForStop = true;
+
+  if (websocket && websocket.readyState === WebSocket.OPEN) {
+    const emptyBlob = new Blob([], { type: "audio/webm" });
+    websocket.send(emptyBlob);
+    statusText.textContent = "Recording stopped. Processing final audio...";
+  }
+
+  if (recorder) {
+    recorder.stop();
+    recorder = null;
+  }
+
+  if (microphone) {
+    microphone.disconnect();
+    microphone = null;
+  }
+
+  if (analyser) {
+    analyser = null;
+  }
+
+  if (audioContext && audioContext.state !== "closed") {
+    try {
+      await audioContext.close();
+    } catch (e) {
+      console.warn("Could not close audio context:", e);
+    }
+    audioContext = null;
+  }
+
+  if (animationFrame) {
+    cancelAnimationFrame(animationFrame);
+    animationFrame = null;
+  }
+
+  if (timerInterval) {
+    clearInterval(timerInterval);
+    timerInterval = null;
+  }
+  timerElement.textContent = "00:00";
+  startTime = null;
+
+  isRecording = false;
+  updateUI();
+}
+
+async function toggleRecording() {
+  if (!isRecording) {
+    if (waitingForStop) {
+      console.log("Waiting for stop, early return");
+      return;
+    }
+    console.log("Connecting to WebSocket");
+    try {
+      if (websocket && websocket.readyState === WebSocket.OPEN) {
+        await startRecording();
+      } else {
+        await setupWebSocket();
+        await startRecording();
+      }
+    } catch (err) {
+      statusText.textContent = "Could not connect to WebSocket or access mic. Aborted.";
+      console.error(err);
+    }
+  } else {
+    console.log("Stopping recording");
+    stopRecording();
+  }
+}
+
+function updateUI() {
+  recordButton.classList.toggle("recording", isRecording);
+  recordButton.disabled = waitingForStop;
+
+  if (waitingForStop) {
+    if (statusText.textContent !== "Recording stopped. Processing final audio...") {
+      statusText.textContent = "Please wait for processing to complete...";
+    }
+  } else if (isRecording) {
+    statusText.textContent = "Recording...";
+  } else {
+    if (
+      statusText.textContent !== "Finished processing audio! Ready to record again." &&
+      statusText.textContent !== "Processing finalized or connection closed."
+    ) {
+      statusText.textContent = "Click to start transcription";
+    }
+  }
+  if (!waitingForStop) {
+    recordButton.disabled = false;
+  }
+}
+
+recordButton.addEventListener("click", toggleRecording);
--- a/whisperlivekit/web/src/dark_mode.svg
+++ b/whisperlivekit/web/src/dark_mode.svg
@@ -0,0 +1 @@
+<svg xmlns="http://www.w3.org/2000/svg" height="24px" viewBox="0 -960 960 960" width="24px" fill="#5f6368"><path d="M480-120q-151 0-255.5-104.5T120-480q0-138 90-239.5T440-838q13-2 23 3.5t16 14.5q6 9 6.5 21t-7.5 23q-17 26-25.5 55t-8.5 61q0 90 63 153t153 63q31 0 61.5-9t54.5-25q11-7 22.5-6.5T819-479q10 5 15.5 15t3.5 24q-14 138-117.5 229T480-120Zm0-80q88 0 158-48.5T740-375q-20 5-40 8t-40 3q-123 0-209.5-86.5T364-660q0-20 3-40t8-40q-78 32-126.5 102T200-480q0 116 82 198t198 82Zm-10-270Z"/></svg>
--- a/whisperlivekit/web/src/light_mode.svg
+++ b/whisperlivekit/web/src/light_mode.svg
@@ -0,0 +1 @@
+<svg xmlns="http://www.w3.org/2000/svg" height="24px" viewBox="0 -960 960 960" width="24px" fill="#5f6368"><path d="M480-360q50 0 85-35t35-85q0-50-35-85t-85-35q-50 0-85 35t-35 85q0 50 35 85t85 35Zm0 80q-83 0-141.5-58.5T280-480q0-83 58.5-141.5T480-680q83 0 141.5 58.5T680-480q0 83-58.5 141.5T480-280ZM80-440q-17 0-28.5-11.5T40-480q0-17 11.5-28.5T80-520h80q17 0 28.5 11.5T200-480q0 17-11.5 28.5T160-440H80Zm720 0q-17 0-28.5-11.5T760-480q0-17 11.5-28.5T800-520h80q17 0 28.5 11.5T920-480q0 17-11.5 28.5T880-440h-80ZM480-760q-17 0-28.5-11.5T440-800v-80q0-17 11.5-28.5T480-920q17 0 28.5 11.5T520-880v80q0 17-11.5 28.5T480-760Zm0 720q-17 0-28.5-11.5T440-80v-80q0-17 11.5-28.5T480-200q17 0 28.5 11.5T520-160v80q0 17-11.5 28.5T480-40ZM226-678l-43-42q-12-11-11.5-28t11.5-29q12-12 29-12t28 12l42 43q11 12 11 28t-11 28q-11 12-27.5 11.5T226-678Zm494 495-42-43q-11-12-11-28.5t11-27.5q11-12 27.5-11.5T734-282l43 42q12 11 11.5 28T777-183q-12 12-29 12t-28-12Zm-42-495q-12-11-11.5-27.5T678-734l42-43q11-12 28-11.5t29 11.5q12 12 12 29t-12 28l-43 42q-12 11-28 11t-28-11ZM183-183q-12-12-12-29t12-28l43-42q12-11 28.5-11t27.5 11q12 11 11.5 27.5T282-226l-42 43q-11 12-28 11.5T183-183Zm297-297Z"/></svg>
--- a/whisperlivekit/web/src/system_mode.svg
+++ b/whisperlivekit/web/src/system_mode.svg
@@ -0,0 +1 @@
+<svg xmlns="http://www.w3.org/2000/svg" height="24px" viewBox="0 -960 960 960" width="24px" fill="#5f6368"><path d="M396-396q-32-32-58.5-67T289-537q-5 14-6.5 28.5T281-480q0 83 58 141t141 58q14 0 28.5-2t28.5-6q-39-22-74-48.5T396-396Zm85 196q-56 0-107-21t-91-61q-40-40-61-91t-21-107q0-51 17-97.5t50-84.5q13-14 32-9.5t27 24.5q21 55 52.5 104t73.5 91q42 42 91 73.5T648-326q20 8 24.5 27t-9.5 32q-38 33-84.5 50T481-200Zm223-192q-16-5-23-20.5t-4-32.5q9-48-6-94.5T621-621q-35-35-80.5-49.5T448-677q-17 3-32-4t-21-23q-6-16 1.5-31t23.5-19q69-15 138 4.5T679-678q51 51 71 120t5 138q-4 17-19 25t-32 3ZM480-840q-17 0-28.5-11.5T440-880v-40q0-17 11.5-28.5T480-960q17 0 28.5 11.5T520-920v40q0 17-11.5 28.5T480-840Zm0 840q-17 0-28.5-11.5T440-40v-40q0-17 11.5-28.5T480-120q17 0 28.5 11.5T520-80v40q0 17-11.5 28.5T480 0Zm255-734q-12-12-12-28.5t12-28.5l28-28q11-11 27.5-11t28.5 11q12 12 12 28.5T819-762l-28 28q-12 12-28 12t-28-12ZM141-141q-12-12-12-28.5t12-28.5l28-28q12-12 28-12t28 12q12 12 12 28.5T225-169l-28 28q-11 11-27.5 11T141-141Zm739-299q-17 0-28.5-11.5T840-480q0-17 11.5-28.5T880-520h40q17 0 28.5 11.5T960-480q0 17-11.5 28.5T920-440h-40Zm-840 0q-17 0-28.5-11.5T0-480q0-17 11.5-28.5T40-520h40q17 0 28.5 11.5T120-480q0 17-11.5 28.5T80-440H40Zm779 299q-12 12-28.5 12T762-141l-28-28q-12-12-12-28t12-28q12-12 28.5-12t28.5 12l28 28q11 11 11 27.5T819-141ZM226-735q-12 12-28.5 12T169-735l-28-28q-11-11-11-27.5t11-28.5q12-12 28.5-12t28.5 12l28 28q12 12 12 28t-12 28Zm170 339Z"/></svg>
--- a/whisperlivekit/web/web_interface.py
+++ b/whisperlivekit/web/web_interface.py
@@ -1,5 +1,6 @@
 import logging
 import importlib.resources as resources
+import base64

 logger = logging.getLogger(__name__)

@@ -10,4 +11,85 @@ def get_web_interface_html():
            return f.read()
    except Exception as e:
        logger.error(f"Error loading web interface HTML: {e}")
-        return "<html><body><h1>Error loading interface</h1></body></html>"
+        return "<html><body><h1>Error loading interface</h1></body></html>"
+
+def get_inline_ui_html():
+    """Returns the complete web interface HTML with all assets embedded in a single call."""
+    try:
+        # Load HTML template
+        with resources.files('whisperlivekit.web').joinpath('live_transcription.html').open('r', encoding='utf-8') as f:
+            html_content = f.read()
+        
+        # Load CSS and embed it
+        with resources.files('whisperlivekit.web').joinpath('live_transcription.css').open('r', encoding='utf-8') as f:
+            css_content = f.read()
+        
+        # Load JS and embed it
+        with resources.files('whisperlivekit.web').joinpath('live_transcription.js').open('r', encoding='utf-8') as f:
+            js_content = f.read()
+        
+        # Load SVG files and convert to data URIs
+        with resources.files('whisperlivekit.web').joinpath('src', 'system_mode.svg').open('r', encoding='utf-8') as f:
+            system_svg = f.read()
+            system_data_uri = f"data:image/svg+xml;base64,{base64.b64encode(system_svg.encode('utf-8')).decode('utf-8')}"
+        
+        with resources.files('whisperlivekit.web').joinpath('src', 'light_mode.svg').open('r', encoding='utf-8') as f:
+            light_svg = f.read()
+            light_data_uri = f"data:image/svg+xml;base64,{base64.b64encode(light_svg.encode('utf-8')).decode('utf-8')}"
+        
+        with resources.files('whisperlivekit.web').joinpath('src', 'dark_mode.svg').open('r', encoding='utf-8') as f:
+            dark_svg = f.read()
+            dark_data_uri = f"data:image/svg+xml;base64,{base64.b64encode(dark_svg.encode('utf-8')).decode('utf-8')}"
+        
+        # Replace external references with embedded content
+        html_content = html_content.replace(
+            '<link rel="stylesheet" href="/web/live_transcription.css" />',
+            f'<style>\n{css_content}\n</style>'
+        )
+        
+        html_content = html_content.replace(
+            '<script src="/web/live_transcription.js"></script>',
+            f'<script>\n{js_content}\n</script>'
+        )
+        
+        # Replace SVG references with data URIs
+        html_content = html_content.replace(
+            '<img src="/web/src/system_mode.svg" alt="" />',
+            f'<img src="{system_data_uri}" alt="" />'
+        )
+        
+        html_content = html_content.replace(
+            '<img src="/web/src/light_mode.svg" alt="" />',
+            f'<img src="{light_data_uri}" alt="" />'
+        )
+        
+        html_content = html_content.replace(
+            '<img src="/web/src/dark_mode.svg" alt="" />',
+            f'<img src="{dark_data_uri}" alt="" />'
+        )
+        
+        return html_content
+        
+    except Exception as e:
+        logger.error(f"Error creating embedded web interface: {e}")
+        return "<html><body><h1>Error loading embedded interface</h1></body></html>"
+
+
+if __name__ == '__main__':
+    
+    from fastapi import FastAPI
+    from fastapi.responses import HTMLResponse
+    import uvicorn
+    from starlette.staticfiles import StaticFiles
+    import pathlib
+    import whisperlivekit.web as webpkg
+    
+    app = FastAPI()    
+    web_dir = pathlib.Path(webpkg.__file__).parent
+    app.mount("/web", StaticFiles(directory=str(web_dir)), name="web")
+    
+    @app.get("/")
+    async def get():
+        return HTMLResponse(get_inline_ui_html())
+
+    uvicorn.run(app=app)
--- a/whisperlivekit/whisper_streaming_custom/online_asr.py
+++ b/whisperlivekit/whisper_streaming_custom/online_asr.py
@@ -122,6 +122,7 @@ class OnlineASRProcessor:
        self.tokenize = tokenize_method
        self.logfile = logfile
        self.confidence_validation = confidence_validation
+        self.global_time_offset = 0.0
        self.init()

        self.buffer_trimming_way, self.buffer_trimming_sec = buffer_trimming
@@ -152,6 +153,21 @@ class OnlineASRProcessor:
        """Append an audio chunk (a numpy array) to the current audio buffer."""
        self.audio_buffer = np.append(self.audio_buffer, audio)

+    def insert_silence(self, silence_duration, offset):
+        """
+        If silences are > 5s, we do a complete context clear. Otherwise, we just insert a small silence and shift the last_attend_frame
+        """
+        # if self.transcript_buffer.buffer:
+        #     self.committed.extend(self.transcript_buffer.buffer)
+        #     self.transcript_buffer.buffer = []
+            
+        if True: #silence_duration < 3: #we want the last audio to be treated to not have a gap. could also be handled in the future in ends_with_silence.
+            gap_silence = np.zeros(int(16000 * silence_duration), dtype=np.int16)
+            self.insert_audio_chunk(gap_silence)
+        else:
+            self.init(offset=silence_duration + offset)
+        self.global_time_offset += silence_duration
+
    def prompt(self) -> Tuple[str, str]:
        """
        Returns a tuple: (prompt, context), where:
@@ -230,6 +246,9 @@ class OnlineASRProcessor:
        logger.debug(
            f"Length of audio buffer now: {len(self.audio_buffer)/self.SAMPLING_RATE:.2f} seconds"
        )
+        if self.global_time_offset:
+            for token in committed_tokens:
+                token = token.with_offset(self.global_time_offset)
        return committed_tokens, current_audio_processed_upto

    def chunk_completed_sentence(self):
@@ -391,128 +410,3 @@ class OnlineASRProcessor:
            start = None
            end = None
        return Transcript(start, end, text, probability=probability)
-
-
-class VACOnlineASRProcessor:
-    """
-    Wraps an OnlineASRProcessor with a Voice Activity Controller (VAC).
-    
-    It receives small chunks of audio, applies VAD (e.g. with Silero),
-    and when the system detects a pause in speech (or end of an utterance)
-    it finalizes the utterance immediately.
-    """
-    SAMPLING_RATE = 16000
-
-    def __init__(self, online_chunk_size: float, *args, **kwargs):
-        self.online_chunk_size = online_chunk_size
-        self.online = OnlineASRProcessor(*args, **kwargs)
-        self.asr = self.online.asr
-        
-        # Load a VAD model (e.g. Silero VAD)
-        import torch
-        model, _ = torch.hub.load(repo_or_dir="snakers4/silero-vad", model="silero_vad")
-        from .silero_vad_iterator import FixedVADIterator
-
-        self.vac = FixedVADIterator(model)
-        self.logfile = self.online.logfile
-        self.last_input_audio_stream_end_time: float = 0.0
-        self.init()
-
-    def init(self):
-        self.online.init()
-        self.vac.reset_states()
-        self.current_online_chunk_buffer_size = 0
-        self.last_input_audio_stream_end_time = self.online.buffer_time_offset
-        self.is_currently_final = False
-        self.status: Optional[str] = None  # "voice" or "nonvoice"
-        self.audio_buffer = np.array([], dtype=np.float32)
-        self.buffer_offset = 0  # in frames
-
-    def get_audio_buffer_end_time(self) -> float:
-        """Returns the absolute end time of the audio processed by the underlying OnlineASRProcessor."""
-        return self.online.get_audio_buffer_end_time()
-
-    def clear_buffer(self):
-        self.buffer_offset += len(self.audio_buffer)
-        self.audio_buffer = np.array([], dtype=np.float32)
-
-    def insert_audio_chunk(self, audio: np.ndarray, audio_stream_end_time: float):
-        """
-        Process an incoming small audio chunk:
-          - run VAD on the chunk,
-          - decide whether to send the audio to the online ASR processor immediately,
-          - and/or to mark the current utterance as finished.
-        """
-        self.last_input_audio_stream_end_time = audio_stream_end_time
-        res = self.vac(audio)
-        self.audio_buffer = np.append(self.audio_buffer, audio)
-
-        if res is not None:
-            # VAD returned a result; adjust the frame number
-            frame = list(res.values())[0] - self.buffer_offset
-            if "start" in res and "end" not in res:
-                self.status = "voice"
-                send_audio = self.audio_buffer[frame:]
-                self.online.init(offset=(frame + self.buffer_offset) / self.SAMPLING_RATE)
-                self.online.insert_audio_chunk(send_audio)
-                self.current_online_chunk_buffer_size += len(send_audio)
-                self.clear_buffer()
-            elif "end" in res and "start" not in res:
-                self.status = "nonvoice"
-                send_audio = self.audio_buffer[:frame]
-                self.online.insert_audio_chunk(send_audio)
-                self.current_online_chunk_buffer_size += len(send_audio)
-                self.is_currently_final = True
-                self.clear_buffer()
-            else:
-                beg = res["start"] - self.buffer_offset
-                end = res["end"] - self.buffer_offset
-                self.status = "nonvoice"
-                send_audio = self.audio_buffer[beg:end]
-                self.online.init(offset=(beg + self.buffer_offset) / self.SAMPLING_RATE)
-                self.online.insert_audio_chunk(send_audio)
-                self.current_online_chunk_buffer_size += len(send_audio)
-                self.is_currently_final = True
-                self.clear_buffer()
-        else:
-            if self.status == "voice":
-                self.online.insert_audio_chunk(self.audio_buffer)
-                self.current_online_chunk_buffer_size += len(self.audio_buffer)
-                self.clear_buffer()
-            else:
-                # Keep 1 second worth of audio in case VAD later detects voice,
-                # but trim to avoid unbounded memory usage.
-                self.buffer_offset += max(0, len(self.audio_buffer) - self.SAMPLING_RATE)
-                self.audio_buffer = self.audio_buffer[-self.SAMPLING_RATE:]
-
-    def process_iter(self) -> Tuple[List[ASRToken], float]:
-        """
-        Depending on the VAD status and the amount of accumulated audio,
-        process the current audio chunk.
-        Returns a tuple: (list of committed ASRToken objects, float representing the audio processed up to time).
-        """
-        if self.is_currently_final:
-            return self.finish()
-        elif self.current_online_chunk_buffer_size > self.SAMPLING_RATE * self.online_chunk_size:
-            self.current_online_chunk_buffer_size = 0
-            return self.online.process_iter()
-        else:
-            logger.debug("No online update, only VAD")
-            return [], self.last_input_audio_stream_end_time
-
-    def finish(self) -> Tuple[List[ASRToken], float]:
-        """
-        Finish processing by flushing any remaining text.
-        Returns a tuple: (list of remaining ASRToken objects, float representing the final audio processed up to time).
-        """
-        result_tokens, processed_upto = self.online.finish()
-        self.current_online_chunk_buffer_size = 0
-        self.is_currently_final = False
-        return result_tokens, processed_upto
-    
-    def get_buffer(self):
-        """
-        Get the unvalidated buffer in string format.
-        """
-        return self.online.concatenate_tokens(self.online.transcript_buffer.buffer)
-
Author	SHA1	Message	Date
Quentin Fuxa	aa44a92a67	add embedded web interface HTML (single-file version with inline CSS/JS/SVG) ### Added - `get_inline_ui_html()`: generates a self-contained version of the web interface, with CSS, JS, and SVG assets inlined directly into the HTML. useful for environments where serving static files is inconvenient or when a single-call UI delivery is preferred.	2025-08-29 21:58:51 +02:00
Quentin Fuxa	01d791470b	add test files	2025-08-29 17:45:32 +02:00
Quentin Fuxa	4a5d5e1f3b	raise Exception when language == auto and task == translation	2025-08-29 17:44:46 +02:00
Quentin Fuxa	583a2ec2e4	highlight Sortformer optional installation	2025-08-27 21:02:25 +02:00
Quentin Fuxa	19765e89e9	remove triton <3 condition	2025-08-27 20:44:39 +02:00
Quentin Fuxa	9895bc83bf	auto detection of language for warmup if not indicated	2025-08-27 20:37:48 +02:00
Quentin Fuxa	ab98c31f16	trim will happen before audio processor	2025-08-27 18:17:11 +02:00
Quentin Fuxa	f9c9c4188a	optional dependencies removed, ask to direct alternative package installations	2025-08-27 18:15:32 +02:00
Quentin Fuxa	c21d2302e7	to 0.2.7	2024-08-24 19:28:00 +02:00
Quentin Fuxa	4ed62e181d	when silences are detected, speaker correction is no more applied	2024-08-24 19:24:00 +02:00
Quentin Fuxa	52a755a08c	indications on how to choose a model	2024-08-24 19:22:00 +02:00
Quentin Fuxa	9a8d3cbd90	improve diarization + silence handling	2024-08-24 19:20:00 +02:00
Quentin Fuxa	b101ce06bd	several users share the same sortformer model instance	2024-08-24 19:18:00 +02:00
Quentin Fuxa	c83fd179a8	improves phase shift correction between transcription and diarization	2024-08-24 19:15:00 +02:00
Quentin Fuxa	5258305745	default diarization backend in now sortformer	2025-08-24 18:32:01 +02:00
Quentin Fuxa	ce781831ee	punctuation is checked in audio-processor's result formatter	2025-08-24 18:32:01 +02:00
Quentin Fuxa	58297daf6d	sortformer diar implementation v0.3	2025-08-24 18:32:01 +02:00
Quentin Fuxa	3393a08f7e	sortformer diar implementation v0.2	2025-08-24 18:32:01 +02:00
Quentin Fuxa	5b2ddeccdb	correct pip installation error in image build	2025-08-22 15:37:46 +02:00
Quentin Fuxa	26cc1072dd	new dockerfile for cpu only. update dockerfile from cuda 12.8 to 12.9	2025-08-22 11:04:35 +02:00
Quentin Fuxa	12973711f6	0.2.6	2025-08-21 14:34:46 +02:00
Quentin Fuxa	909ac9dd41	speaker -1 are no more sent in websocket - no buffer when their is a silence	2025-08-21 14:09:02 +02:00
Quentin Fuxa	d94a07d417	default model is now base. default backend simulstreaming	2025-08-21 11:55:36 +02:00
Quentin Fuxa	b32dd8bfc4	Align backend and frontend time handling	2025-08-21 10:33:15 +02:00
Quentin Fuxa	9feb0e597b	remove VACOnlineASRProcessor backend possibility	2025-08-20 20:57:43 +02:00
Quentin Fuxa	9dab84a573	update front	2025-08-20 20:15:38 +02:00
Quentin Fuxa	d089c7fce0	.html to .html + .css + .js	2025-08-20 20:00:31 +02:00
Quentin Fuxa	253a080df5	diart diarization handles pauses/silences thanks to offset	2025-08-19 21:12:55 +02:00
Quentin Fuxa	0c6e4b2aee	sortformer diar implementation v0.1	2025-08-19 19:48:51 +02:00
Quentin Fuxa	e14bbde77d	sortformer diar implementation v0	2025-08-19 17:02:55 +02:00
Quentin Fuxa	7496163467	rename diart backend	2025-08-19 15:02:27 +02:00
Quentin Fuxa	696a94d1ce	1rst sortformer backend implementation	2025-08-19 15:02:17 +02:00
Quentin Fuxa	2699b0974c	Fix simulstreaming imports	2025-08-19 14:43:54 +02:00
Quentin Fuxa	90c0250ba4	update optional dependencies	2025-08-19 09:36:59 +02:00
Quentin Fuxa	eb96153ffd	new vac parameters	2025-08-17 22:26:28 +02:00
Quentin Fuxa	47e3eb9b5b	Update README.md	2025-08-17 09:55:03 +02:00
Quentin Fuxa	b8b07adeef	--vac to --no-vac	2025-08-17 09:44:26 +02:00
Quentin Fuxa	d0e9e37ef6	simulstreaming: cumulative_time_offset to keep timestamps correct when audio > 30s	2025-08-17 09:33:47 +02:00
Quentin Fuxa	820f92d8cb	audio_max_len to 30 -> 20, ffmpeg timeout 5 -> 20	2025-08-17 09:32:08 +02:00
Quentin Fuxa	e42523af84	VAC activated by default	2025-08-17 01:29:34 +02:00
Quentin Fuxa	e2184d5e06	better handle silences when VAC + correct offset issue with whisperstreaming backend	2025-08-17 01:27:07 +02:00
Quentin Fuxa	7fe0353260	vac model is loaded in TranscriptionEngine, and by default	2025-08-17 00:34:25 +02:00
Quentin Fuxa	0f2eba507e	use with_offset to add no audio offset to tokens	2025-08-17 00:33:24 +02:00
Quentin Fuxa	55e08474f3	recycle backend in simulstreaming thanks to new remove hooks function	2025-08-16 23:06:16 +02:00
Quentin Fuxa	28bdc52e1d	VAC before doing transcription and diarization. V0	2025-08-16 23:04:21 +02:00
Quentin Fuxa	e4221fa6c3	Merge branch 'main' of https://github.com/QuentinFuxa/whisper_streaming_web	2025-08-15 23:04:05 +02:00
Quentin Fuxa	1652db9a2d	Use distinct backend models for simulstreaming and add --preloaded_model_count to preload them	2025-08-15 23:03:55 +02:00
Quentin Fuxa	601f17653a	Update CONTRIBUTING.md	2025-08-13 21:59:32 +02:00
Quentin Fuxa	7718190fcd	Update CONTRIBUTING.md	2025-08-13 21:59:00 +02:00
				`@@ -0,0 +1 @@`
				`<svg xmlns="http://www.w3.org/2000/svg" height="24px" viewBox="0 -960 960 960" width="24px" fill="#5f6368"><path d="M480-120q-151 0-255.5-104.5T120-480q0-138 90-239.5T440-838q13-2 23 3.5t16 14.5q6 9 6.5 21t-7.5 23q-17 26-25.5 55t-8.5 61q0 90 63 153t153 63q31 0 61.5-9t54.5-25q11-7 22.5-6.5T819-479q10 5 15.5 15t3.5 24q-14 138-117.5 229T480-120Zm0-80q88 0 158-48.5T740-375q-20 5-40 8t-40 3q-123 0-209.5-86.5T364-660q0-20 3-40t8-40q-78 32-126.5 102T200-480q0 116 82 198t198 82Zm-10-270Z"/></svg>`
				`@@ -0,0 +1 @@`
				<svg xmlns="http://www.w3.org/2000/svg" height="24px" viewBox="0 -960 960 960" width="24px" fill="#5f6368"><path d="M480-360q50 0 85-35t35-85q0-50-35-85t-85-35q-50 0-85 35t-35 85q0 50 35 85t85 35Zm0 80q-83 0-141.5-58.5T280-480q0-83 58.5-141.5T480-680q83 0 141.5 58.5T680-480q0 83-58.5 141.5T480-280ZM80-440q-17 0-28.5-11.5T40-480q0-17 11.5-28.5T80-520h80q17 0 28.5 11.5T200-480q0 17-11.5 28.5T160-440H80Zm720 0q-17 0-28.5-11.5T760-480q0-17 11.5-28.5T800-520h80q17 0 28.5 11.5T920-480q0 17-11.5 28.5T880-440h-80ZM480-760q-17 0-28.5-11.5T440-800v-80q0-17 11.5-28.5T480-920q17 0 28.5 11.5T520-880v80q0 17-11.5 28.5T480-760Zm0 720q-17 0-28.5-11.5T440-80v-80q0-17 11.5-28.5T480-200q17 0 28.5 11.5T520-160v80q0 17-11.5 28.5T480-40ZM226-678l-43-42q-12-11-11.5-28t11.5-29q12-12 29-12t28 12l42 43q11 12 11 28t-11 28q-11 12-27.5 11.5T226-678Zm494 495-42-43q-11-12-11-28.5t11-27.5q11-12 27.5-11.5T734-282l43 42q12 11 11.5 28T777-183q-12 12-29 12t-28-12Zm-42-495q-12-11-11.5-27.5T678-734l42-43q11-12 28-11.5t29 11.5q12 12 12 29t-12 28l-43 42q-12 11-28 11t-28-11ZM183-183q-12-12-12-29t12-28l43-42q12-11 28.5-11t27.5 11q12 11 11.5 27.5T282-226l-42 43q-11 12-28 11.5T183-183Zm297-297Z"/></svg>