moneychatbot

Running

App Files Files Community

hadadrjt commited on Sep 4

Commit

408c946

0 Parent(s):

SearchGPT: Initial.

Browse files

Signed-off-by: Hadad <[email protected]>

Files changed (19) hide show

.gitattributes +35 -0
Dockerfile +22 -0
README.md +75 -0
app.py +34 -0
assets/css/__init__.py +8 -0
assets/css/reasoning.py +31 -0
config.py +267 -0
requirements.txt +2 -0
src/client/__init__.py +8 -0
src/client/openai_client.py +17 -0
src/core/__init__.py +12 -0
src/core/web_configuration.py +13 -0
src/core/web_loader.py +188 -0
src/engine/__init__.py +8 -0
src/engine/browser_engine.py +88 -0
src/processor/__init__.py +8 -0
src/processor/message_processor.py +237 -0
src/tools/__init__.py +8 -0
src/tools/tool_manager.py +50 -0

.gitattributes ADDED Viewed

	@@ -0,0 +1,35 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text

Dockerfile ADDED Viewed

	@@ -0,0 +1,22 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+# Use a specific container image for the app
+FROM python:latest
+# Set the main working directory inside the container
+WORKDIR /app
+# Copy all files into the container
+COPY . .
+# Install all dependencies
+RUN pip install -r requirements.txt
+# Open the port so the app can be accessed
+EXPOSE 7860
+# Start the app
+CMD ["python", "app.py"]

README.md ADDED Viewed

	@@ -0,0 +1,75 @@

+---
+title: SearchGPT
+short_description: ChatGPT with real-time web search & URL reading capability
+license: apache-2.0
+emoji: ⚡
+colorFrom: blue
+colorTo: yellow
+sdk: docker
+app_port: 7860
+pinned: false
+# Used to promote this Hugging Face Space
+models:
+- hadadrjt/JARVIS
+- agentica-org/DeepCoder-14B-Preview
+- agentica-org/DeepSWE-Preview
+- fka/awesome-chatgpt-prompts
+- black-forest-labs/FLUX.1-Kontext-dev
+- ChatDOC/OCRFlux-3B
+- deepseek-ai/DeepSeek-R1
+- deepseek-ai/DeepSeek-R1-0528
+- deepseek-ai/DeepSeek-R1-Distill-Llama-70B
+- deepseek-ai/DeepSeek-R1-Distill-Qwen-32B
+- deepseek-ai/DeepSeek-R1-0528-Qwen3-8B
+- deepseek-ai/DeepSeek-V3-0324
+- google/gemma-3-1b-it
+- google/gemma-3-27b-it
+- google/gemma-3-4b-it
+- google/gemma-3n-E4B-it
+- google/gemma-3n-E4B-it-litert-preview
+- google/medsiglip-448
+- kyutai/tts-1.6b-en_fr
+- meta-llama/Llama-3.1-8B-Instruct
+- meta-llama/Llama-3.2-3B-Instruct
+- meta-llama/Llama-3.3-70B-Instruct
+- meta-llama/Llama-4-Maverick-17B-128E-Instruct
+- meta-llama/Llama-4-Scout-17B-16E-Instruct
+- microsoft/Phi-4-mini-instruct
+- mistralai/Devstral-Small-2505
+- mistralai/Mistral-Small-3.1-24B-Instruct-2503
+- openai/webgpt_comparisons
+- openai/whisper-large-v3-turbo
+- openai/gpt-oss-120b
+- openai/gpt-oss-20b
+- Qwen/QwQ-32B
+- Qwen/Qwen2.5-VL-32B-Instruct
+- Qwen/Qwen2.5-VL-3B-Instruct
+- Qwen/Qwen2.5-VL-72B-Instruct
+- Qwen/Qwen3-235B-A22B
+- THUDM/GLM-4.1V-9B-Thinking
+- tngtech/DeepSeek-TNG-R1T2-Chimera
+- moonshotai/Kimi-K2-Instruct
+- Qwen/Qwen3-235B-A22B-Instruct-2507
+- Qwen/Qwen3-Coder-480B-A35B-Instruct
+- Qwen/Qwen3-235B-A22B-Thinking-2507
+- zai-org/GLM-4.5
+- zai-org/GLM-4.5-Air
+- zai-org/GLM-4.5V
+- deepseek-ai/DeepSeek-V3.1
+- deepseek-ai/DeepSeek-V3.1-Base
+- microsoft/VibeVoice-1.5B
+- xai-org/grok-2
+- Qwen/Qwen-Image-Edit
+- ByteDance-Seed/Seed-OSS-36B-Instruct
+- google/gemma-3-270m
+- google/gemma-3-270m-it
+- openbmb/MiniCPM-V-4_5
+- tencent/Hunyuan-MT-7B
+- meituan-longcat/LongCat-Flash-Chat
+- Phr00t/WAN2.2-14B-Rapid-AllInOne
+- apple/FastVLM-0.5B
+- stepfun-ai/Step-Audio-2-mini
+# Used to promote this Hugging Face Space
+datasets:
+- fka/awesome-chatgpt-prompts
+---

app.py ADDED Viewed

	@@ -0,0 +1,34 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from src.processor.message_processor import process_user_request
+from config import DESCRIPTION
+import gradio as gr
+with gr.Blocks(fill_height=True, fill_width=True) as app:
+    with gr.Sidebar(): gr.HTML(DESCRIPTION)
+    gr.ChatInterface(
+        fn=process_user_request,
+        chatbot=gr.Chatbot(
+            label="SearchGPT | GPT-4.1 (Nano)",
+            type="messages",
+            show_copy_button=True,
+            scale=1
+        ),
+        type="messages",
+        examples=[
+            ["What is UltimaX Intelligence"],
+            ["https://wikipedia.org/wiki/Artificial_intelligence Read and summarize that"],
+            ["What's the latest AI development in 2025?"],
+            ["OpenAI GPT-5 vs DeepSeek V3.1"]
+        ],
+        cache_examples=False,
+        show_api=False
+    )
+app.launch(
+    server_name="0.0.0.0",
+    pwa=True
+)

assets/css/__init__.py ADDED Viewed

	@@ -0,0 +1,8 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .reasoning import styles
+__all__ = ['styles']

assets/css/reasoning.py ADDED Viewed

	@@ -0,0 +1,31 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+def styles(reasoning: str, expanded: bool = False) -> str:
+    open_attr = "open" if expanded else ""
+    emoji = "&#129504;"
+    return f"""
+<details {open_attr} style="
+    font-family: 'Segoe UI', Tahoma, Geneva, Verdana, sans-serif;
+">
+  <summary style="
+    font-weight: 700;
+    font-size: 14px !important;
+    cursor: pointer;
+    user-select: none;
+  ">
+    {emoji} Reasoning
+  </summary>
+  <div style="
+    margin-top: 6px;
+    padding-top: 6px;
+    font-size: 10px !important;
+    line-height: 1.7;
+    letter-spacing: 0.02em;
+  ">
+    {reasoning}
+  </div>
+</details>
+"""

config.py ADDED Viewed

	@@ -0,0 +1,267 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+#OPENAI_API_BASE_URL  # Endpoint. Not here -> Hugging Face Spaces secrets
+#OPENAI_API_KEY       # API Key.  Not here -> Hugging Face Spaces secrets
+MODEL = "gpt-4.1-nano"
+SEARXNG_ENDPOINT = "https://searx.stream/search"  # See the endpoint list at https://searx.space
+BAIDU_ENDPOINT   = "https://www.baidu.com/s"
+READER_ENDPOINT  = "https://r.jina.ai/"
+REQUEST_TIMEOUT  = 300  # 5 minute
+INSTRUCTIONS = """
+You are ChatGPT with advanced real-time web search, content extraction, and summarization capabilities.
+Your objective is to provide the most accurate, comprehensive, and professionally structured responses to user queries.
+Always use web search to gather relevant information before responding unless the question is purely factual and does not require external sources.
+Search workflow :
+1. Perform a web search using available engines (Google, Bing, Baidu) to retrieve highly relevant results
+2. Select up to 10 top results based on relevance, credibility, and content depth
+3. For each selected URL, fetch the full content using the read_url function
+4. Extract key information, critical data, and insights
+5. Collect all URLs encountered in search results and content extraction
+6. Provide a structured summary in English, professional, concise, and precise
+7. Include citations for each URL used, in the format [Source title](URL)
+8. If information is ambiguous, incomplete, or contradictory, clearly state it
+9. Ensure your response is readable, logically organized, and free of emoji, dashes, or unnecessary symbols
+"""
+CONTENT_EXTRACTION = """
+<system>
+- Analyze the retrieved content in detail
+- Identify all critical facts, arguments, statistics, and relevant data
+- Collect all URLs, hyperlinks, references, and citations mentioned in the content
+- Evaluate credibility of sources, highlight potential biases or conflicts
+- Produce a structured, professional, and comprehensive summary
+- Emphasize clarity, accuracy, and logical flow
+- Include all discovered URLs in the final summary as [Source title](URL)
+- Mark any uncertainties, contradictions, or missing information clearly
+</system>
+"""
+SEARCH_SELECTION = """
+<system>
+- For each search result, fetch the full content using read_url
+- Extract key information, main arguments, data points, and statistics
+- Capture every URL present in the content or references
+- Create a professional, structured summary in English
+- List each source at the end of the summary in the format [Source title](link)
+- Identify ambiguities or gaps in information
+- Ensure clarity, completeness, and high information density
+</system>
+"""
+DESCRIPTION = """
+<b>SearchGPT</b> is <b>ChatGPT</b> with real-time web search capabilities and the ability to read content directly from a URL.
+<br><br>
+This Space implements an agent-based system with <b><a href="https://www.gradio.app" target="_blank">Gradio</a></b>. It is integrated with
+<b><a href="https://docs.searxng.org" target="_blank">SearXNG</a></b>, which is then converted into a script tool or function for native execution.
+<br><br>
+The agent mode is inspired by the <b><a href="https://openwebui.com/t/hadad/deep_research" target="_blank">Deep Research</a></b> from
+<b><a href="https://docs.openwebui.com" target="_blank">OpenWebUI</a></b> tools script.
+<br><br>
+The <b>Deep Research</b> feature is also available on the primary Spaces of <b><a href="https://umint-openwebui.hf.space"
+target="_blank">UltimaX Intelligence</a></b>.
+<br><br>
+Please consider reading the <b><a href="https://huggingface.co/spaces/umint/ai/discussions/37#68b55209c51ca52ed299db4c"
+target="_blank">Terms of Use and Consequences of Violation</a></b> if you wish to proceed to the main Spaces.
+<br><br>
+<b>Like this project? Feel free to buy me a <a href="https://ko-fi.com/hadad" target="_blank">coffee</a></b>.
+"""
+OS = [
+    "Windows NT 10.0; Win64; x64",
+    "Macintosh; Intel Mac OS X 10_15_7",
+    "X11; Linux x86_64",
+    "Windows NT 11.0; Win64; x64",
+    "Macintosh; Intel Mac OS X 11_6_2"
+]
+OCTETS = [
+     1,   2,   3,   4,   5,   8,  12,  13,  14,  15,
+    16,  17,  18,  19,  20,  23,  24,  34,  35,  36,
+    37,  38,  39,  40,  41,  42,  43,  44,  45,  46,
+    47,  48,  49,  50,  51,  52,  53,  54,  55,  56,
+    57,  58,  59,  60,  61,  62,  63,  64,  65,  66,
+    67,  68,  69,  70,  71,  72,  73,  74,  75,  76,
+    77,  78,  79,  80,  81,  82,  83,  84,  85,  86,
+    87,  88,  89,  90,  91,  92,  93,  94,  95,  96,
+    97,  98,  99, 100, 101, 102, 103, 104, 105, 106,
+   107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
+   117, 118, 119, 120, 121, 122, 123, 124, 125, 126,
+   128, 129, 130, 131, 132, 133, 134, 135, 136, 137,
+   138, 139, 140, 141, 142, 143, 144, 145, 146, 147,
+   148, 149, 150, 151, 152, 153, 154, 155, 156, 157,
+   158, 159, 160, 161, 162, 163, 164, 165, 166, 167,
+   168, 170, 171, 172, 173, 174, 175, 176, 177, 178,
+   179, 180, 181, 182, 183, 184, 185, 186, 187, 188,
+   189, 190, 191, 192, 193, 194, 195, 196, 197, 198,
+   199, 200, 201, 202, 203, 204, 205, 206, 207, 208,
+   209, 210, 211, 212, 213, 214, 215, 216, 217, 218,
+   219, 220, 221, 222, 223
+]
+BROWSERS = [
+    "Chrome",
+    "Firefox",
+    "Safari",
+    "Edge",
+    "Opera"
+]
+CHROME_VERSIONS = [
+    "120.0.0.0",
+    "119.0.0.0",
+    "118.0.0.0",
+    "117.0.0.0",
+    "116.0.0.0"
+]
+FIREFOX_VERSIONS = [
+    "121.0",
+    "120.0",
+    "119.0",
+    "118.0",
+    "117.0"
+]
+SAFARI_VERSIONS = [
+    "17.1",
+    "17.0",
+    "16.6",
+    "16.5",
+    "16.4",
+]
+EDGE_VERSIONS = [
+    "120.0.2210.91",
+    "119.0.2151.97",
+    "118.0.2088.76",
+    "117.0.2045.60",
+    "116.0.1938.81"
+]
+DOMAINS = [
+    "google.com",
+    "bing.com",
+    "yahoo.com",
+    "duckduckgo.com",
+    "baidu.com",
+    "yandex.com",
+    "facebook.com",
+    "twitter.com",
+    "linkedin.com",
+    "reddit.com",
+    "youtube.com",
+    "wikipedia.org",
+    "amazon.com",
+    "github.com",
+    "stackoverflow.com",
+    "medium.com",
+    "quora.com",
+    "pinterest.com",
+    "instagram.com",
+    "tumblr.com"
+]
+PROTOCOLS = [
+    "https://",
+    "https://www."
+]
+SEARCH_ENGINES = [
+    "https://www.google.com/search?q=",
+    "https://www.bing.com/search?q=",
+    "https://search.yahoo.com/search?p=",
+    "https://duckduckgo.com/?q=",
+    "https://www.baidu.com/s?wd=",
+    "https://yandex.com/search/?text=",
+    "https://www.google.co.uk/search?q=",
+    "https://www.google.ca/search?q=",
+    "https://www.google.com.au/search?q=",
+    "https://www.google.de/search?q=",
+    "https://www.google.fr/search?q=",
+    "https://www.google.co.jp/search?q=",
+    "https://www.google.com.br/search?q=",
+    "https://www.google.co.in/search?q=",
+    "https://www.google.ru/search?q=",
+    "https://www.google.it/search?q="
+]
+KEYWORDS = [
+    "news",
+    "weather",
+    "sports",
+    "technology",
+    "science",
+    "health",
+    "finance",
+    "entertainment",
+    "travel",
+    "food",
+    "education",
+    "business",
+    "politics",
+    "culture",
+    "history",
+    "music",
+    "movies",
+    "games",
+    "books",
+    "art"
+]
+COUNTRIES = [
+    "US", "GB", "CA", "AU", "DE", "FR", "JP", "BR", "IN", "RU",
+    "IT", "ES", "MX", "NL", "SE", "NO", "DK", "FI", "PL", "TR",
+    "KR", "SG", "HK", "TW", "TH", "ID", "MY", "PH", "VN", "AR",
+    "CL", "CO", "PE", "VE", "EG", "ZA", "NG", "KE", "MA", "DZ",
+    "TN", "IL", "AE", "SA", "QA", "KW", "BH", "OM", "JO", "LB"
+]
+LANGUAGES = [
+    "en-US", "en-GB", "en-CA", "en-AU", "de-DE", "fr-FR", "ja-JP",
+    "pt-BR", "hi-IN", "ru-RU", "it-IT", "es-ES", "es-MX", "nl-NL",
+    "sv-SE", "no-NO", "da-DK", "fi-FI", "pl-PL", "tr-TR", "ko-KR",
+    "zh-CN", "zh-TW", "th-TH", "id-ID", "ms-MY", "fil-PH", "vi-VN",
+    "es-AR", "es-CL", "es-CO", "es-PE", "es-VE", "ar-EG", "en-ZA",
+    "en-NG", "sw-KE", "ar-MA", "ar-DZ", "ar-TN", "he-IL", "ar-AE",
+    "ar-SA", "ar-QA", "ar-KW", "ar-BH", "ar-OM", "ar-JO", "ar-LB"
+]
+TIMEZONES = [
+    "America/New_York",
+    "America/Chicago",
+    "America/Los_Angeles",
+    "America/Denver",
+    "Europe/London",
+    "Europe/Paris",
+    "Europe/Berlin",
+    "Europe/Moscow",
+    "Asia/Tokyo",
+    "Asia/Shanghai",
+    "Asia/Hong_Kong",
+    "Asia/Singapore",
+    "Asia/Seoul",
+    "Asia/Mumbai",
+    "Asia/Dubai",
+    "Australia/Sydney",
+    "Australia/Melbourne",
+    "America/Toronto",
+    "America/Vancouver",
+    "America/Mexico_City",
+    "America/Sao_Paulo",
+    "America/Buenos_Aires",
+    "Africa/Cairo",
+    "Africa/Johannesburg",
+    "Africa/Lagos",
+    "Africa/Nairobi",
+    "Pacific/Auckland",
+    "Pacific/Honolulu"
+]

requirements.txt ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ gradio[oauth,mcp]
2	+ openai

src/client/__init__.py ADDED Viewed

	@@ -0,0 +1,8 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .openai_client import initialize_client
+__all__ = ['initialize_client']

src/client/openai_client.py ADDED Viewed

	@@ -0,0 +1,17 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+import os
+from openai import OpenAI
+def initialize_client():
+    try:
+        client = OpenAI(
+            base_url=os.getenv("OPENAI_API_BASE_URL"),
+            api_key=os.getenv("OPENAI_API_KEY")
+        )
+        return client, None
+    except Exception as initialization_error:
+        return None, f"Failed to initialize client: {str(initialization_error)}"

src/core/__init__.py ADDED Viewed

	@@ -0,0 +1,12 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .web_loader import WebLoader
+from .web_configuration import WebConfiguration
+__all__ = [
+    'WebLoader',
+    'WebConfiguration'
+]

src/core/web_configuration.py ADDED Viewed

	@@ -0,0 +1,13 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from config import SEARXNG_ENDPOINT, BAIDU_ENDPOINT, READER_ENDPOINT, REQUEST_TIMEOUT
+class WebConfiguration:
+    def __init__(self):
+        self.searxng_endpoint = SEARXNG_ENDPOINT
+        self.baidu_endpoint = BAIDU_ENDPOINT
+        self.content_reader_api = READER_ENDPOINT
+        self.request_timeout = REQUEST_TIMEOUT

src/core/web_loader.py ADDED Viewed

	@@ -0,0 +1,188 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+import random
+import threading
+import time
+from collections import deque
+from config import (
+    OS,
+    OCTETS,
+    BROWSERS,
+    CHROME_VERSIONS,
+    FIREFOX_VERSIONS,
+    SAFARI_VERSIONS,
+    EDGE_VERSIONS,
+    DOMAINS,
+    PROTOCOLS,
+    SEARCH_ENGINES,
+    KEYWORDS,
+    COUNTRIES,
+    LANGUAGES,
+    TIMEZONES
+)
+class WebLoader:
+    def __init__(self):
+        self.ipv4_pool = deque(maxlen=1000)
+        self.ipv6_pool = deque(maxlen=1000)
+        self.user_agent_pool = deque(maxlen=500)
+        self.origin_pool = deque(maxlen=500)
+        self.referrer_pool = deque(maxlen=500)
+        self.location_pool = deque(maxlen=500)
+        self.lock = threading.Lock()
+        self.running = True
+    def generate_ipv4(self):
+        while len(self.ipv4_pool) < 1000 and self.running:
+            octet = random.choice(OCTETS)
+            ip = f"{octet}.{random.randint(0, 255)}.{random.randint(0, 255)}.{random.randint(1, 254)}"
+            with self.lock:
+                self.ipv4_pool.append(ip)
+            time.sleep(0.001)
+    def generate_ipv6(self):
+        while len(self.ipv6_pool) < 1000 and self.running:
+            segments = []
+            for _ in range(8):
+                segments.append(f"{random.randint(0, 65535):04x}")
+            ip = ":".join(segments)
+            with self.lock:
+                self.ipv6_pool.append(ip)
+            time.sleep(0.001)
+    def generate_user_agents(self):
+        os_list = OS
+        browsers = BROWSERS
+        chrome_versions = CHROME_VERSIONS
+        firefox_versions = FIREFOX_VERSIONS
+        safari_versions = SAFARI_VERSIONS
+        edge_versions = EDGE_VERSIONS
+        while len(self.user_agent_pool) < 500 and self.running:
+            browser = random.choice(browsers)
+            os_string = random.choice(os_list)
+            if browser == "Chrome":
+                version = random.choice(chrome_versions)
+                ua = f"Mozilla/5.0 ({os_string}) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/{version} Safari/537.36"
+            elif browser == "Firefox":
+                version = random.choice(firefox_versions)
+                ua = f"Mozilla/5.0 ({os_string}) Gecko/20100101 Firefox/{version}"
+            elif browser == "Safari":
+                version = random.choice(safari_versions)
+                webkit_version = f"{600 + random.randint(0, 15)}.{random.randint(1, 9)}.{random.randint(1, 20)}"
+                ua = f"Mozilla/5.0 ({os_string}) AppleWebKit/{webkit_version} (KHTML, like Gecko) Version/{version} Safari/{webkit_version}"
+            elif browser == "Edge":
+                version = random.choice(edge_versions)
+                ua = f"Mozilla/5.0 ({os_string}) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/{version.split('.')[0]}.0.0.0 Safari/537.36 Edg/{version}"
+            else:
+                version = f"{random.randint(70, 100)}.0.{random.randint(3000, 5000)}.{random.randint(50, 150)}"
+                ua = f"Mozilla/5.0 ({os_string}) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/{version} Safari/537.36 OPR/{random.randint(80, 106)}.0.0.0"
+            with self.lock:
+                self.user_agent_pool.append(ua)
+            time.sleep(0.002)
+    def generate_origins(self):
+        domains = DOMAINS
+        protocols = PROTOCOLS
+        while len(self.origin_pool) < 500 and self.running:
+            protocol = random.choice(protocols)
+            domain = random.choice(domains)
+            origin = f"{protocol}{domain}"
+            with self.lock:
+                self.origin_pool.append(origin)
+            time.sleep(0.002)
+    def generate_referrers(self):
+        search_engines = SEARCH_ENGINES
+        keywords = KEYWORDS
+        while len(self.referrer_pool) < 500 and self.running:
+            engine = random.choice(search_engines)
+            keyword = random.choice(keywords)
+            referrer = f"{engine}{keyword}"
+            with self.lock:
+                self.referrer_pool.append(referrer)
+            time.sleep(0.002)
+    def generate_locations(self):
+        countries = COUNTRIES
+        languages = LANGUAGES
+        timezones = TIMEZONES
+        while len(self.location_pool) < 500 and self.running:
+            country = random.choice(countries)
+            language = random.choice(languages)
+            timezone = random.choice(timezones)
+            location = {
+                "country": country,
+                "language": language,
+                "timezone": timezone
+            }
+            with self.lock:
+                self.location_pool.append(location)
+            time.sleep(0.002)
+    def get_ipv4(self):
+        with self.lock:
+            if self.ipv4_pool:
+                return self.ipv4_pool[random.randint(0, len(self.ipv4_pool) - 1)]
+        return f"{random.randint(1, 223)}.{random.randint(0, 255)}.{random.randint(0, 255)}.{random.randint(1, 254)}"
+    def get_ipv6(self):
+        with self.lock:
+            if self.ipv6_pool:
+                return self.ipv6_pool[random.randint(0, len(self.ipv6_pool) - 1)]
+        segments = [f"{random.randint(0, 65535):04x}" for _ in range(8)]
+        return ":".join(segments)
+    def get_user_agent(self):
+        with self.lock:
+            if self.user_agent_pool:
+                return self.user_agent_pool[random.randint(0, len(self.user_agent_pool) - 1)]
+        return "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
+    def get_origin(self):
+        with self.lock:
+            if self.origin_pool:
+                return self.origin_pool[random.randint(0, len(self.origin_pool) - 1)]
+        return "https://www.google.com"
+    def get_referrer(self):
+        with self.lock:
+            if self.referrer_pool:
+                return self.referrer_pool[random.randint(0, len(self.referrer_pool) - 1)]
+        return "https://www.google.com/search?q=search"
+    def get_location(self):
+        with self.lock:
+            if self.location_pool:
+                return self.location_pool[random.randint(0, len(self.location_pool) - 1)]
+        return {
+            "country": "US",
+            "language": "en-US",
+            "timezone": "America/New_York"
+        }
+    def start_engine(self):
+        threads = [
+            threading.Thread(target=self.generate_ipv4, daemon=True),
+            threading.Thread(target=self.generate_ipv6, daemon=True),
+            threading.Thread(target=self.generate_user_agents, daemon=True),
+            threading.Thread(target=self.generate_origins, daemon=True),
+            threading.Thread(target=self.generate_referrers, daemon=True),
+            threading.Thread(target=self.generate_locations, daemon=True)
+        ]
+        for thread in threads:
+            thread.start()
+    def stop(self):
+        self.running = False
+web_loader = WebLoader()
+web_loader.start_engine()

src/engine/__init__.py ADDED Viewed

	@@ -0,0 +1,8 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .browser_engine import BrowserEngine
+__all__ = ['BrowserEngine']

src/engine/browser_engine.py ADDED Viewed

	@@ -0,0 +1,88 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+import requests
+from config import CONTENT_EXTRACTION, SEARCH_SELECTION
+from src.core.web_loader import web_loader
+class BrowserEngine:
+    def __init__(self, configuration):
+        self.config = configuration
+    def generate_headers(self):
+        ipv4 = web_loader.get_ipv4()
+        ipv6 = web_loader.get_ipv6()
+        user_agent = web_loader.get_user_agent()
+        origin = web_loader.get_origin()
+        referrer = web_loader.get_referrer()
+        location = web_loader.get_location()
+        return {
+            "User-Agent": user_agent,
+            "X-Forwarded-For": f"{ipv4}, {ipv6}",
+            "X-Real-IP": ipv4,
+            "X-Originating-IP": ipv4,
+            "X-Remote-IP": ipv4,
+            "X-Remote-Addr": ipv4,
+            "X-Client-IP": ipv4,
+            "X-Forwarded-Host": origin.replace("https://", "").replace("http://", ""),
+            "Origin": origin,
+            "Referer": referrer,
+            "Accept-Language": f"{location['language']},en;q=0.9",
+            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
+            "Accept-Encoding": "gzip, deflate, br",
+            "DNT": "1",
+            "Connection": "keep-alive",
+            "Upgrade-Insecure-Requests": "1",
+            "Sec-Fetch-Dest": "document",
+            "Sec-Fetch-Mode": "navigate",
+            "Sec-Fetch-Site": "cross-site",
+            "Sec-Fetch-User": "?1",
+            "Cache-Control": "max-age=0",
+            "X-Country": location['country'],
+            "X-Timezone": location['timezone']
+        }
+    def extract_page_content(self, target_url: str) -> str:
+        try:
+            headers = self.generate_headers()
+            payload = {
+                "url": target_url
+            }
+            request_response = requests.post(
+                self.config.content_reader_api,
+                data=payload,
+                headers=headers,
+                timeout=self.config.request_timeout,
+            )
+            request_response.raise_for_status()
+            extracted_content = request_response.text
+            return f"{extracted_content}{CONTENT_EXTRACTION}"
+        except Exception as error:
+            return f"Error reading URL: {str(error)}"
+    def perform_search(self, search_query: str, search_provider: str = "google") -> str:
+        try:
+            headers = self.generate_headers()
+            if search_provider == "baidu":
+                full_url = f"{self.config.content_reader_api}{self.config.baidu_endpoint}?wd={requests.utils.quote(search_query)}"
+                headers["X-Target-Selector"] = "#content_left"
+            else:
+                provider_prefix = "!go" if search_provider == "google" else "!bi"
+                encoded_query = requests.utils.quote(f"{provider_prefix} {search_query}")
+                full_url = f"{self.config.content_reader_api}{self.config.searxng_endpoint}?q={encoded_query}"
+                headers["X-Target-Selector"] = "#urls"
+            search_response = requests.get(
+                full_url,
+                headers=headers,
+                timeout=self.config.request_timeout
+            )
+            search_response.raise_for_status()
+            search_results = search_response.text
+            return f"{search_results}{SEARCH_SELECTION}"
+        except Exception as error:
+            return f"Error during search: {str(error)}"

src/processor/__init__.py ADDED Viewed

	@@ -0,0 +1,8 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .message_processor import process_user_request
+__all__ = ['process_user_request']

src/processor/message_processor.py ADDED Viewed

	@@ -0,0 +1,237 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+import json
+import traceback
+from openai import OpenAI
+from config import MODEL, INSTRUCTIONS
+from src.core.web_configuration import WebConfiguration
+from src.engine.browser_engine import BrowserEngine
+from src.tools.tool_manager import construct_tool_definitions
+from src.client.openai_client import initialize_client
+from assets.css.reasoning import styles
+def setup_response(system_instruction, conversation_history, user_input):
+    history = []
+    if system_instruction:
+        history.append({"role": "system", "content": system_instruction})
+    if isinstance(conversation_history, list):
+        for history_item in conversation_history:
+            message_role = history_item.get("role")
+            message_content = history_item.get("content")
+            if message_role in ("user", "assistant") and isinstance(message_content, str):
+                history.append({"role": message_role, "content": message_content})
+    if isinstance(user_input, str) and user_input.strip():
+        history.append({"role": "user", "content": user_input})
+    return history
+def extract_tool_parameters(raw_parameters, fallback_engine="google"):
+    try:
+        parsed_params = json.loads(raw_parameters or "{}")
+        if "engine" in parsed_params and parsed_params["engine"] not in ["google", "bing", "baidu"]:
+            parsed_params["engine"] = fallback_engine
+        if "engine" not in parsed_params:
+            parsed_params["engine"] = fallback_engine
+        return parsed_params, None
+    except Exception as parse_error:
+        return None, f"Invalid tool arguments: {str(parse_error)}"
+def assistant_response(response_message):
+    extracted_tool_calls = []
+    if getattr(response_message, "tool_calls", None):
+        for tool_call in response_message.tool_calls:
+            extracted_tool_calls.append(
+                {
+                    "id": tool_call.id,
+                    "type": "function",
+                    "function": {
+                        "name": tool_call.function.name,
+                        "arguments": tool_call.function.arguments
+                    }
+                }
+            )
+    return {
+        "role": "assistant",
+        "content": response_message.content or "",
+        "tool_calls": extracted_tool_calls if extracted_tool_calls else None
+    }
+def invoke_tool_function(search_engine, function_name, function_params):
+    if function_name == "web_search":
+        return search_engine.perform_search(
+            search_query=function_params.get("query", ""),
+            search_provider=function_params.get("engine", "google")
+        )
+    if function_name == "read_url":
+        return search_engine.extract_page_content(
+            target_url=function_params.get("url", "")
+        )
+    return f"Unknown tool: {function_name}"
+def generate_response(server, model_name, conversation_messages, tool_definitions):
+    response_generator = ""
+    try:
+        response = server.chat.completions.create(
+            model=model_name,
+            messages=conversation_messages,
+            tools=tool_definitions,
+            tool_choice="none",
+            temperature=1.0,
+            stream=True
+        )
+        for data in response:
+            try:
+                raw_data = data.choices[0].delta.content or ""
+            except Exception:
+                raw_data = ""
+            if raw_data:
+                response_generator += raw_data
+                yield response_generator
+        yield response_generator
+    except Exception as response_error:
+        response_generator += f"\nError: {str(response_error)}\n"
+        response_generator += traceback.format_exc()
+        yield response_generator
+def process_tool_interactions(server, model_name, conversation_messages, tool_definitions, search_engine):
+    maximum_iterations = 4
+    logs_generator = ""
+    for iteration_index in range(maximum_iterations):
+        try:
+            model_response = server.chat.completions.create(
+                model=model_name,
+                messages=conversation_messages,
+                tools=tool_definitions,
+                tool_choice="auto",
+                temperature=0.7
+            )
+        except Exception:
+            return conversation_messages, logs_generator
+        response_choice = model_response.choices[0]
+        assistant_message = response_choice.message
+        formatted_assistant_message = assistant_response(assistant_message)
+        conversation_messages.append(
+            {
+                "role": formatted_assistant_message["role"],
+                "content": formatted_assistant_message["content"],
+                "tool_calls": formatted_assistant_message["tool_calls"]
+            }
+        )
+        pending_tool_calls = assistant_message.tool_calls or []
+        if not pending_tool_calls:
+            return conversation_messages, logs_generator
+        for tool_invocation in pending_tool_calls:
+            tool_name = tool_invocation.function.name
+            tool_arguments_raw = tool_invocation.function.arguments
+            extracted_arguments, extraction_error = extract_tool_parameters(tool_arguments_raw)
+            if extraction_error:
+                log_content = f"Tool: {tool_name}<br>Status: Failed<br>Error: {extraction_error}"
+                logs_generator = styles(log_content, expanded=True)
+                yield logs_generator
+                tool_execution_result = extraction_error
+            else:
+                log_content = f"Tool: {tool_name}<br>Status: Executing<br>Parameters: {json.dumps(extracted_arguments, indent=2).replace(' ', '&nbsp;').replace(chr(10), '<br>')}"
+                logs_generator = styles(log_content, expanded=True)
+                yield logs_generator
+                tool_execution_result = invoke_tool_function(
+                    search_engine,
+                    tool_name,
+                    extracted_arguments
+                )
+                result_preview = tool_execution_result[:500] + "..." if len(tool_execution_result) > 500 else tool_execution_result
+                log_content = f"Tool: {tool_name}<br>Status: Completed<br>Parameters: {json.dumps(extracted_arguments, indent=2).replace(' ', '&nbsp;').replace(chr(10), '<br>')}<br>Result Preview: {result_preview.replace(chr(10), '<br>')}"
+                logs_generator = styles(log_content, expanded=False)
+                yield logs_generator
+            conversation_messages.append(
+                {
+                    "role": "tool",
+                    "tool_call_id": tool_invocation.id,
+                    "name": tool_name,
+                    "content": tool_execution_result
+                }
+            )
+    return conversation_messages, logs_generator
+def process_user_request(user_message, chat_history):
+    if not isinstance(user_message, str) or not user_message.strip():
+        yield []
+        return
+    output_content = ""
+    try:
+        server, client_initialization_error = initialize_client()
+        if client_initialization_error:
+            output_content = client_initialization_error
+            yield output_content
+            return
+        search_configuration = WebConfiguration()
+        search_engine_instance = BrowserEngine(search_configuration)
+        available_tools = construct_tool_definitions()
+        conversation_messages = setup_response(
+            INSTRUCTIONS,
+            chat_history,
+            user_message
+        )
+        tool_response = ""
+        for tool_update in process_tool_interactions(
+            server=server,
+            model_name=MODEL,
+            conversation_messages=conversation_messages,
+            tool_definitions=available_tools,
+            search_engine=search_engine_instance
+        ):
+            if isinstance(tool_update, str):
+                tool_response = tool_update
+                yield tool_response
+            else:
+                conversation_messages = tool_update[0]
+                tool_response = tool_update[1]
+        if tool_response:
+            yield tool_response + "\n\n"
+        final_response_generator = generate_response(
+            server=server,
+            model_name=MODEL,
+            conversation_messages=conversation_messages,
+            tool_definitions=available_tools
+        )
+        for response_final in final_response_generator:
+            if tool_response:
+                yield tool_response + "\n\n" + response_final
+            else:
+                yield response_final
+    except Exception as processing_error:
+        output_content += f"\nError: {str(processing_error)}\n"
+        output_content += traceback.format_exc()
+        yield output_content

src/tools/__init__.py ADDED Viewed

	@@ -0,0 +1,8 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+from .tool_manager import construct_tool_definitions
+__all__ = ['construct_tool_definitions']

src/tools/tool_manager.py ADDED Viewed

	@@ -0,0 +1,50 @@

+#
+# SPDX-FileCopyrightText: Hadad <[email protected]>
+# SPDX-License-Identifier: Apache-2.0
+#
+def construct_tool_definitions():
+    return [
+        {
+            "type": "function",
+            "function": {
+                "name": "web_search",
+                "description": "Perform a web search via SearXNG (Google or Bing) or Baidu.",
+                "parameters": {
+                    "type": "object",
+                    "properties": {
+                        "query": {
+                            "type": "string"
+                        },
+                        "engine": {
+                            "type": "string",
+                            "enum": [
+                                "google",
+                                "bing",
+                                "baidu"
+                            ],
+                            "default": "google",
+                        },
+                    },
+                    "required": ["query"],
+                },
+            },
+        },
+        {
+            "type": "function",
+            "function": {
+                "name": "read_url",
+                "description": "Fetch and extract main content from a URL.",
+                "parameters": {
+                    "type": "object",
+                    "properties": {
+                        "url": {
+                            "type": "string",
+                            "format": "uri"
+                        },
+                    },
+                    "required": ["url"],
+                },
+            },
+        }
+    ]