Spaces:

fastrtc
/

talk-to-claude-gradio

Running

App Files Files Community

freddyaboulton HF staff commited on 6 days ago

Commit

19b72df

verified ·

1 Parent(s): 0c83ad6

Upload folder using huggingface_hub

Browse files

Files changed (4) hide show

README.md +9 -6
app.py +136 -0
index.html +335 -0
requirements.txt +6 -0

README.md CHANGED Viewed

@@ -1,12 +1,15 @@
 ---
-title: Talk To Claude Gradio
-emoji: ⚡
-colorFrom: yellow
-colorTo: green
 sdk: gradio
-sdk_version: 5.16.1
 app_file: app.py
 pinned: false
 ---
-Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

 ---
+title: Talk to Claude
+emoji: 👨‍🦰
+colorFrom: purple
+colorTo: red
 sdk: gradio
+sdk_version: 5.16.0
 app_file: app.py
 pinned: false
+license: mit
+short_description: Talk to Anthropic's Claude
+tags: [webrtc, websocket, gradio, secret|TWILIO_ACCOUNT_SID, secret|TWILIO_AUTH_TOKEN, secret|GROQ_API_KEY, secret|ANTHROPIC_API_KEY, secret|ELEVENLABS_API_KEY]
 ---
+Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

app.py ADDED Viewed

	@@ -0,0 +1,136 @@

+import json
+import os
+from pathlib import Path
+import anthropic
+import gradio as gr
+import numpy as np
+from dotenv import load_dotenv
+from elevenlabs import ElevenLabs
+from fastapi import FastAPI
+from fastapi.responses import HTMLResponse, StreamingResponse
+from fastrtc import (
+    AdditionalOutputs,
+    ReplyOnPause,
+    Stream,
+    get_tts_model,
+    get_twilio_turn_credentials,
+)
+from fastrtc.utils import audio_to_bytes
+from gradio.utils import get_space
+from groq import Groq
+from pydantic import BaseModel
+load_dotenv()
+groq_client = Groq()
+claude_client = anthropic.Anthropic()
+tts_client = ElevenLabs(api_key=os.environ["ELEVENLABS_API_KEY"])
+curr_dir = Path(__file__).parent
+tts_model = get_tts_model()
+def response(
+    audio: tuple[int, np.ndarray],
+    chatbot: list[dict] | None = None,
+):
+    chatbot = chatbot or []
+    messages = [{"role": d["role"], "content": d["content"]} for d in chatbot]
+    prompt = groq_client.audio.transcriptions.create(
+        file=("audio-file.mp3", audio_to_bytes(audio)),
+        model="whisper-large-v3-turbo",
+        response_format="verbose_json",
+    ).text
+    print("prompt", prompt)
+    chatbot.append({"role": "user", "content": prompt})
+    yield AdditionalOutputs(chatbot)
+    messages.append({"role": "user", "content": prompt})
+    response = claude_client.messages.create(
+        model="claude-3-5-haiku-20241022",
+        max_tokens=512,
+        messages=messages,  # type: ignore
+    )
+    response_text = " ".join(
+        block.text  # type: ignore
+        for block in response.content
+        if getattr(block, "type", None) == "text"
+    )
+    chatbot.append({"role": "assistant", "content": response_text})
+    import time
+    start = time.time()
+    print("starting tts", start)
+    for i, chunk in enumerate(tts_model.stream_tts_sync(response_text)):
+        print("chunk", i, time.time() - start)
+        yield chunk
+    print("finished tts", time.time() - start)
+    yield AdditionalOutputs(chatbot)
+chatbot = gr.Chatbot(type="messages")
+stream = Stream(
+    modality="audio",
+    mode="send-receive",
+    handler=ReplyOnPause(response),
+    additional_outputs_handler=lambda a, b: b,
+    additional_inputs=[chatbot],
+    additional_outputs=[chatbot],
+    rtc_configuration=get_twilio_turn_credentials() if get_space() else None,
+    concurrency_limit=20 if get_space() else None,
+)
+class Message(BaseModel):
+    role: str
+    content: str
+class InputData(BaseModel):
+    webrtc_id: str
+    chatbot: list[Message]
+app = FastAPI()
+stream.mount(app)
+@app.get("/")
+async def _():
+    rtc_config = get_twilio_turn_credentials() if get_space() else None
+    html_content = (curr_dir / "index.html").read_text()
+    html_content = html_content.replace("__RTC_CONFIGURATION__", json.dumps(rtc_config))
+    return HTMLResponse(content=html_content, status_code=200)
+@app.post("/input_hook")
+async def _(body: InputData):
+    stream.set_input(body.webrtc_id, body.model_dump()["chatbot"])
+    return {"status": "ok"}
+@app.get("/outputs")
+def _(webrtc_id: str):
+    async def output_stream():
+        async for output in stream.output_stream(webrtc_id):
+            chatbot = output.args[0]
+            if len(chatbot) > 1:
+                yield f"event: output\ndata: {json.dumps(chatbot[-2])}\n\n"
+                yield f"event: output\ndata: {json.dumps(chatbot[-1])}\n\n"
+    return StreamingResponse(output_stream(), media_type="text/event-stream")
+if __name__ == "__main__":
+    import os
+    if (mode := os.getenv("MODE")) == "UI":
+        stream.ui.launch(server_port=7860)
+    elif mode == "PHONE":
+        stream.fastphone(host="0.0.0.0", port=7860)
+    else:
+        import uvicorn
+        uvicorn.run(app, host="0.0.0.0", port=7860)

index.html ADDED Viewed

	@@ -0,0 +1,335 @@

+<!DOCTYPE html>
+<html lang="en">
+<head>
+    <meta charset="UTF-8">
+    <meta name="viewport" content="width=device-width, initial-scale=1.0">
+    <title>RetroChat Audio</title>
+    <style>
+        body {
+            font-family: monospace;
+            background-color: #1a1a1a;
+            color: #00ff00;
+            margin: 0;
+            padding: 20px;
+            height: 100vh;
+            box-sizing: border-box;
+        }
+        .container {
+            display: grid;
+            grid-template-columns: 1fr 1fr;
+            gap: 20px;
+            height: calc(100% - 100px);
+            margin-bottom: 20px;
+        }
+        .visualization-container {
+            border: 2px solid #00ff00;
+            padding: 20px;
+            display: flex;
+            flex-direction: column;
+            align-items: center;
+            position: relative;
+        }
+        #visualizer {
+            width: 100%;
+            height: 100%;
+            background-color: #000;
+        }
+        .chat-container {
+            border: 2px solid #00ff00;
+            padding: 20px;
+            display: flex;
+            flex-direction: column;
+            height: 100%;
+            box-sizing: border-box;
+        }
+        .chat-messages {
+            flex-grow: 1;
+            overflow-y: auto;
+            margin-bottom: 20px;
+            padding: 10px;
+            border: 1px solid #00ff00;
+        }
+        .message {
+            margin-bottom: 10px;
+            padding: 8px;
+            border-radius: 4px;
+        }
+        .message.user {
+            background-color: #003300;
+        }
+        .message.assistant {
+            background-color: #002200;
+        }
+        .controls {
+            text-align: center;
+        }
+        button {
+            background-color: #000;
+            color: #00ff00;
+            border: 2px solid #00ff00;
+            padding: 10px 20px;
+            font-family: monospace;
+            font-size: 16px;
+            cursor: pointer;
+            transition: all 0.3s;
+        }
+        button:hover {
+            background-color: #00ff00;
+            color: #000;
+        }
+        #audio-output {
+            display: none;
+        }
+        /* Retro CRT effect */
+        .crt-overlay {
+            position: absolute;
+            top: 0;
+            left: 0;
+            width: 100%;
+            height: 100%;
+            background: repeating-linear-gradient(0deg,
+                    rgba(0, 255, 0, 0.03),
+                    rgba(0, 255, 0, 0.03) 1px,
+                    transparent 1px,
+                    transparent 2px);
+            pointer-events: none;
+        }
+    </style>
+</head>
+<body>
+    <div class="container">
+        <div class="visualization-container">
+            <canvas id="visualizer"></canvas>
+            <div class="crt-overlay"></div>
+        </div>
+        <div class="chat-container">
+            <div class="chat-messages" id="chat-messages"></div>
+        </div>
+    </div>
+    <div class="controls">
+        <button id="start-button">Start</button>
+    </div>
+    <audio id="audio-output"></audio>
+    <script>
+        let audioContext;
+        let analyser;
+        let dataArray;
+        let animationId;
+        let chatHistory = [];
+        let peerConnection;
+        let webrtc_id;
+        const visualizer = document.getElementById('visualizer');
+        const ctx = visualizer.getContext('2d');
+        const audioOutput = document.getElementById('audio-output');
+        const startButton = document.getElementById('start-button');
+        const chatMessages = document.getElementById('chat-messages');
+        // Set canvas size
+        function resizeCanvas() {
+            visualizer.width = visualizer.offsetWidth;
+            visualizer.height = visualizer.offsetHeight;
+        }
+        window.addEventListener('resize', resizeCanvas);
+        resizeCanvas();
+        // Initialize WebRTC
+        async function setupWebRTC() {
+            const config = __RTC_CONFIGURATION__;
+            peerConnection = new RTCPeerConnection(config);
+            try {
+                const stream = await navigator.mediaDevices.getUserMedia({
+                    audio: true
+                });
+                stream.getTracks().forEach(track => {
+                    peerConnection.addTrack(track, stream);
+                });
+                // Audio visualization will be set up when we receive the output stream
+                // Handle incoming audio
+                peerConnection.addEventListener('track', (evt) => {
+                    if (audioOutput && audioOutput.srcObject !== evt.streams[0]) {
+                        audioOutput.srcObject = evt.streams[0];
+                        audioOutput.play();
+                        // Set up audio visualization on the output stream
+                        audioContext = new AudioContext();
+                        analyser = audioContext.createAnalyser();
+                        const source = audioContext.createMediaStreamSource(evt.streams[0]);
+                        source.connect(analyser);
+                        analyser.fftSize = 2048;
+                        dataArray = new Uint8Array(analyser.frequencyBinCount);
+                    }
+                });
+                // Create data channel for messages
+                const dataChannel = peerConnection.createDataChannel('text');
+                dataChannel.onmessage = handleMessage;
+                // Create and send offer
+                const offer = await peerConnection.createOffer();
+                await peerConnection.setLocalDescription(offer);
+                await new Promise((resolve) => {
+                    if (peerConnection.iceGatheringState === "complete") {
+                        resolve();
+                    } else {
+                        const checkState = () => {
+                            if (peerConnection.iceGatheringState === "complete") {
+                                peerConnection.removeEventListener("icegatheringstatechange", checkState);
+                                resolve();
+                            }
+                        };
+                        peerConnection.addEventListener("icegatheringstatechange", checkState);
+                    }
+                });
+                webrtc_id = Math.random().toString(36).substring(7);
+                const response = await fetch('/webrtc/offer', {
+                    method: 'POST',
+                    headers: { 'Content-Type': 'application/json' },
+                    body: JSON.stringify({
+                        sdp: peerConnection.localDescription.sdp,
+                        type: peerConnection.localDescription.type,
+                        webrtc_id: webrtc_id
+                    })
+                });
+                const serverResponse = await response.json();
+                await peerConnection.setRemoteDescription(serverResponse);
+                // Start visualization
+                draw();
+                // create event stream to receive messages from /output
+                const eventSource = new EventSource('/outputs?webrtc_id=' + webrtc_id);
+                eventSource.addEventListener("output", (event) => {
+                    const eventJson = JSON.parse(event.data);
+                    addMessage(eventJson.role, eventJson.content);
+                });
+            } catch (err) {
+                console.error('Error setting up WebRTC:', err);
+            }
+        }
+        function handleMessage(event) {
+            const eventJson = JSON.parse(event.data);
+            if (eventJson.type === "send_input") {
+                fetch('/input_hook', {
+                    method: 'POST',
+                    headers: {
+                        'Content-Type': 'application/json',
+                    },
+                    body: JSON.stringify({
+                        webrtc_id: webrtc_id,
+                        chatbot: chatHistory
+                    })
+                });
+            }
+        }
+        function addMessage(role, content) {
+            const messageDiv = document.createElement('div');
+            messageDiv.classList.add('message', role);
+            messageDiv.textContent = content;
+            chatMessages.appendChild(messageDiv);
+            chatMessages.scrollTop = chatMessages.scrollHeight;
+            chatHistory.push({ role, content });
+        }
+        function draw() {
+            animationId = requestAnimationFrame(draw);
+            analyser.getByteTimeDomainData(dataArray);
+            ctx.fillStyle = 'rgb(0, 0, 0)';
+            ctx.fillRect(0, 0, visualizer.width, visualizer.height);
+            ctx.lineWidth = 2;
+            ctx.strokeStyle = 'rgb(0, 255, 0)';
+            ctx.beginPath();
+            const sliceWidth = visualizer.width / dataArray.length;
+            let x = 0;
+            for (let i = 0; i < dataArray.length; i++) {
+                const v = dataArray[i] / 128.0;
+                const y = v * visualizer.height / 2;
+                if (i === 0) {
+                    ctx.moveTo(x, y);
+                } else {
+                    ctx.lineTo(x, y);
+                }
+                x += sliceWidth;
+            }
+            ctx.lineTo(visualizer.width, visualizer.height / 2);
+            ctx.stroke();
+        }
+        function stop() {
+            if (peerConnection) {
+                if (peerConnection.getTransceivers) {
+                    peerConnection.getTransceivers().forEach(transceiver => {
+                        if (transceiver.stop) {
+                            transceiver.stop();
+                        }
+                    });
+                }
+                if (peerConnection.getSenders) {
+                    peerConnection.getSenders().forEach(sender => {
+                        if (sender.track && sender.track.stop) sender.track.stop();
+                    });
+                }
+                setTimeout(() => {
+                    peerConnection.close();
+                }, 500);
+            }
+            if (animationId) {
+                cancelAnimationFrame(animationId);
+            }
+            if (audioContext) {
+                audioContext.close();
+            }
+        }
+        startButton.addEventListener('click', () => {
+            if (startButton.textContent === 'Start') {
+                setupWebRTC();
+                startButton.textContent = 'Stop';
+            } else {
+                stop();
+                startButton.textContent = 'Start';
+            }
+        });
+    </script>
+</body>
+</html>

requirements.txt ADDED Viewed

	@@ -0,0 +1,6 @@

+fastrtc[vad, tts]
+elevenlabs
+groq
+anthropic
+twilio
+python-dotenv