chyundunovDatamonsters
diff --git a/‎AgentQnA/docker_compose/amd/gpu/rocm/README.md
Lines changed: 301 additions & 68 deletions b/‎AgentQnA/docker_compose/amd/gpu/rocm/README.md
Lines changed: 301 additions & 68 deletions
diff --git a/‎AgentQnA/docker_compose/amd/gpu/rocm/compose.yaml
Lines changed: 49 additions & 22 deletions b/‎AgentQnA/docker_compose/amd/gpu/rocm/compose.yaml
Lines changed: 49 additions & 22 deletions
diff --git a/‎AgentQnA/docker_compose/amd/gpu/rocm/compose_vllm.yaml
Lines changed: 128 additions & 0 deletions b/‎AgentQnA/docker_compose/amd/gpu/rocm/compose_vllm.yaml
Lines changed: 128 additions & 0 deletions
diff --git a/‎AgentQnA/docker_compose/amd/gpu/rocm/launch_agent_service_tgi_rocm.sh
Lines changed: 67 additions & 27 deletions b/‎AgentQnA/docker_compose/amd/gpu/rocm/launch_agent_service_tgi_rocm.sh
Lines changed: 67 additions & 27 deletions
@@ -1,26 +1,24 @@
-# Copyright (C) 2024 Intel Corporation
-# SPDX-License-Identifier: Apache-2.0
+# Copyright (C) 2025 Advanced Micro Devices, Inc.
 
 services:
-  agent-tgi-server:
-    image: ${AGENTQNA_TGI_IMAGE}
-    container_name: agent-tgi-server
+  tgi-service:
+    image: ghcr.io/huggingface/text-generation-inference:3.0.0-rocm
+    container_name: tgi-service
     ports:
-      - "${AGENTQNA_TGI_SERVICE_PORT-8085}:80"
+      - "${TGI_SERVICE_PORT-8085}:80"
     volumes:
-      - ${HF_CACHE_DIR:-/var/opea/agent-service/}:/data
+      - "${MODEL_CACHE:-./data}:/data"
     environment:
       no_proxy: ${no_proxy}
       http_proxy: ${http_proxy}
       https_proxy: ${https_proxy}
-      TGI_LLM_ENDPOINT: "http://${HOST_IP}:${AGENTQNA_TGI_SERVICE_PORT}"
+      TGI_LLM_ENDPOINT: "http://${ip_address}:${TGI_SERVICE_PORT}"
       HUGGING_FACE_HUB_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
       HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
-    shm_size: 1g
+    shm_size: 32g
     devices:
       - /dev/kfd:/dev/kfd
-      - /dev/dri/${AGENTQNA_CARD_ID}:/dev/dri/${AGENTQNA_CARD_ID}
-      - /dev/dri/${AGENTQNA_RENDER_ID}:/dev/dri/${AGENTQNA_RENDER_ID}
+      - /dev/dri:/dev/dri
     cap_add:
       - SYS_PTRACE
     group_add:
@@ -34,14 +32,14 @@ services:
     image: opea/agent:latest
     container_name: rag-agent-endpoint
     volumes:
-      # - ${WORKDIR}/GenAIExamples/AgentQnA/docker_image_build/GenAIComps/comps/agent/langchain/:/home/user/comps/agent/langchain/
-      - ${TOOLSET_PATH}:/home/user/tools/
+      - "${TOOLSET_PATH}:/home/user/tools/"
     ports:
-      - "9095:9095"
+      - "${WORKER_RAG_AGENT_PORT:-9095}:9095"
     ipc: host
     environment:
       ip_address: ${ip_address}
       strategy: rag_agent_llama
+      with_memory: false
       recursion_limit: ${recursion_limit_worker}
       llm_engine: tgi
       HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
@@ -61,29 +59,57 @@ services:
       LANGCHAIN_PROJECT: "opea-worker-agent-service"
       port: 9095
 
+  worker-sql-agent:
+    image: opea/agent:latest
+    container_name: sql-agent-endpoint
+    volumes:
+      - "${WORKDIR}/tests/Chinook_Sqlite.sqlite:/home/user/chinook-db/Chinook_Sqlite.sqlite:rw"
+    ports:
+      - "${WORKER_SQL_AGENT_PORT:-9096}:9096"
+    ipc: host
+    environment:
+      ip_address: ${ip_address}
+      strategy: sql_agent_llama
+      with_memory: false
+      db_name: ${db_name}
+      db_path: ${db_path}
+      use_hints: false
+      recursion_limit: ${recursion_limit_worker}
+      llm_engine: vllm
+      HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      llm_endpoint_url: ${LLM_ENDPOINT_URL}
+      model: ${LLM_MODEL_ID}
+      temperature: ${temperature}
+      max_new_tokens: ${max_new_tokens}
+      stream: false
+      require_human_feedback: false
+      no_proxy: ${no_proxy}
+      http_proxy: ${http_proxy}
+      https_proxy: ${https_proxy}
+      port: 9096
+
   supervisor-react-agent:
     image: opea/agent:latest
     container_name: react-agent-endpoint
     depends_on:
-      - agent-tgi-server
       - worker-rag-agent
     volumes:
-      # - ${WORKDIR}/GenAIExamples/AgentQnA/docker_image_build/GenAIComps/comps/agent/langchain/:/home/user/comps/agent/langchain/
-      - ${TOOLSET_PATH}:/home/user/tools/
+      - "${TOOLSET_PATH}:/home/user/tools/"
     ports:
-      - "${AGENTQNA_FRONTEND_PORT}:9090"
+      - "${SUPERVISOR_REACT_AGENT_PORT:-9090}:9090"
     ipc: host
     environment:
       ip_address: ${ip_address}
-      strategy: react_langgraph
+      strategy: react_llama
+      with_memory: true
       recursion_limit: ${recursion_limit_supervisor}
       llm_engine: tgi
       HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
       llm_endpoint_url: ${LLM_ENDPOINT_URL}
       model: ${LLM_MODEL_ID}
       temperature: ${temperature}
       max_new_tokens: ${max_new_tokens}
-      stream: false
+      stream: true
       tools: /home/user/tools/supervisor_agent_tools.yaml
       require_human_feedback: false
       no_proxy: ${no_proxy}
@@ -92,6 +118,7 @@ services:
       LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
       LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
       LANGCHAIN_PROJECT: "opea-supervisor-agent-service"
-      CRAG_SERVER: $CRAG_SERVER
-      WORKER_AGENT_URL: $WORKER_AGENT_URL
+      CRAG_SERVER: ${CRAG_SERVER}
+      WORKER_AGENT_URL: ${WORKER_AGENT_URL}
+      SQL_AGENT_URL: ${SQL_AGENT_URL}
       port: 9090
@@ -0,0 +1,128 @@
+# Copyright (C) 2025 Advanced Micro Devices, Inc.
+
+services:
+  vllm-service:
+    image: ${REGISTRY:-opea}/vllm-rocm:${TAG:-latest}
+    container_name: vllm-service
+    ports:
+      - "${VLLM_SERVICE_PORT:-8081}:8011"
+    environment:
+      no_proxy: ${no_proxy}
+      http_proxy: ${http_proxy}
+      https_proxy: ${https_proxy}
+      HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      HF_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      HF_HUB_DISABLE_PROGRESS_BARS: 1
+      HF_HUB_ENABLE_HF_TRANSFER: 0
+      WILM_USE_TRITON_FLASH_ATTENTION: 0
+      PYTORCH_JIT: 0
+    volumes:
+      - "${MODEL_CACHE:-./data}:/data"
+    shm_size: 20G
+    devices:
+      - /dev/kfd:/dev/kfd
+      - /dev/dri/:/dev/dri/
+    cap_add:
+      - SYS_PTRACE
+    group_add:
+      - video
+    security_opt:
+      - seccomp:unconfined
+      - apparmor=unconfined
+    command: "--model ${VLLM_LLM_MODEL_ID} --swap-space 16 --disable-log-requests --dtype float16 --tensor-parallel-size 4 --host 0.0.0.0 --port 8011 --num-scheduler-steps 1 --distributed-executor-backend \"mp\""
+    ipc: host
+
+  worker-rag-agent:
+    image: opea/agent:latest
+    container_name: rag-agent-endpoint
+    volumes:
+      - ${TOOLSET_PATH}:/home/user/tools/
+    ports:
+      - "${WORKER_RAG_AGENT_PORT:-9095}:9095"
+    ipc: host
+    environment:
+      ip_address: ${ip_address}
+      strategy: rag_agent_llama
+      with_memory: false
+      recursion_limit: ${recursion_limit_worker}
+      llm_engine: vllm
+      HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      llm_endpoint_url: ${LLM_ENDPOINT_URL}
+      model: ${LLM_MODEL_ID}
+      temperature: ${temperature}
+      max_new_tokens: ${max_new_tokens}
+      stream: false
+      tools: /home/user/tools/worker_agent_tools.yaml
+      require_human_feedback: false
+      RETRIEVAL_TOOL_URL: ${RETRIEVAL_TOOL_URL}
+      no_proxy: ${no_proxy}
+      http_proxy: ${http_proxy}
+      https_proxy: ${https_proxy}
+      LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
+      LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
+      LANGCHAIN_PROJECT: "opea-worker-agent-service"
+      port: 9095
+
+  worker-sql-agent:
+    image: opea/agent:latest
+    container_name: sql-agent-endpoint
+    volumes:
+      - "${WORKDIR}/tests/Chinook_Sqlite.sqlite:/home/user/chinook-db/Chinook_Sqlite.sqlite:rw"
+    ports:
+      - "${WORKER_SQL_AGENT_PORT:-9096}:9096"
+    ipc: host
+    environment:
+      ip_address: ${ip_address}
+      strategy: sql_agent_llama
+      with_memory: false
+      db_name: ${db_name}
+      db_path: ${db_path}
+      use_hints: false
+      recursion_limit: ${recursion_limit_worker}
+      llm_engine: vllm
+      HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      llm_endpoint_url: ${LLM_ENDPOINT_URL}
+      model: ${LLM_MODEL_ID}
+      temperature: ${temperature}
+      max_new_tokens: ${max_new_tokens}
+      stream: false
+      require_human_feedback: false
+      no_proxy: ${no_proxy}
+      http_proxy: ${http_proxy}
+      https_proxy: ${https_proxy}
+      port: 9096
+
+  supervisor-react-agent:
+    image: opea/agent:latest
+    container_name: react-agent-endpoint
+    depends_on:
+      - worker-rag-agent
+    volumes:
+      - ${TOOLSET_PATH}:/home/user/tools/
+    ports:
+      - "${SUPERVISOR_REACT_AGENT_PORT:-9090}:9090"
+    ipc: host
+    environment:
+      ip_address: ${ip_address}
+      strategy: react_llama
+      with_memory: true
+      recursion_limit: ${recursion_limit_supervisor}
+      llm_engine: vllm
+      HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
+      llm_endpoint_url: ${LLM_ENDPOINT_URL}
+      model: ${LLM_MODEL_ID}
+      temperature: ${temperature}
+      max_new_tokens: ${max_new_tokens}
+      stream: true
+      tools: /home/user/tools/supervisor_agent_tools.yaml
+      require_human_feedback: false
+      no_proxy: ${no_proxy}
+      http_proxy: ${http_proxy}
+      https_proxy: ${https_proxy}
+      LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
+      LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
+      LANGCHAIN_PROJECT: "opea-supervisor-agent-service"
+      CRAG_SERVER: ${CRAG_SERVER}
+      WORKER_AGENT_URL: ${WORKER_AGENT_URL}
+      SQL_AGENT_URL: ${SQL_AGENT_URL}
+      port: 9090
@@ -1,47 +1,87 @@
 # Copyright (C) 2024 Advanced Micro Devices, Inc.
 # SPDX-License-Identifier: Apache-2.0
 
-WORKPATH=$(dirname "$PWD")/..
+# Before start script:
+# export host_ip="your_host_ip_or_host_name"
+# export HUGGINGFACEHUB_API_TOKEN="your_huggingface_api_token"
+# export LANGCHAIN_API_KEY="your_langchain_api_key"
+# export LANGCHAIN_TRACING_V2=""
+
+# Set server hostname or IP address
 export ip_address=${host_ip}
-export HUGGINGFACEHUB_API_TOKEN=${your_hf_api_token}
-export AGENTQNA_TGI_IMAGE=ghcr.io/huggingface/text-generation-inference:2.4.1-rocm
-export AGENTQNA_TGI_SERVICE_PORT="8085"
 
-# LLM related environment variables
-export AGENTQNA_CARD_ID="card1"
-export AGENTQNA_RENDER_ID="renderD136"
-export HF_CACHE_DIR=${HF_CACHE_DIR}
-ls $HF_CACHE_DIR
-export LLM_MODEL_ID="meta-llama/Meta-Llama-3-8B-Instruct"
-#export NUM_SHARDS=4
-export LLM_ENDPOINT_URL="http://${ip_address}:${AGENTQNA_TGI_SERVICE_PORT}"
+# Set services IP ports
+export TGI_SERVICE_PORT="18110"
+export WORKER_RAG_AGENT_PORT="18111"
+export WORKER_SQL_AGENT_PORT="18112"
+export SUPERVISOR_REACT_AGENT_PORT="18113"
+export CRAG_SERVER_PORT="18114"
+
+export WORKPATH=$(dirname "$PWD")
+export WORKDIR=${WORKPATH}/../../../
+export HUGGINGFACEHUB_API_TOKEN=${HUGGINGFACEHUB_API_TOKEN}
+export LLM_MODEL_ID="Intel/neural-chat-7b-v3-3"
+export HF_CACHE_DIR="./data"
+export MODEL_CACHE="./data"
+export TOOLSET_PATH=${WORKPATH}/../../../tools/
+export recursion_limit_worker=12
+export LLM_ENDPOINT_URL=http://${ip_address}:${TGI_SERVICE_PORT}
 export temperature=0.01
 export max_new_tokens=512
-
-# agent related environment variables
-export AGENTQNA_WORKER_AGENT_SERVICE_PORT="9095"
-export TOOLSET_PATH=/home/huggingface/datamonsters/amd-opea/GenAIExamples/AgentQnA/tools/
-echo "TOOLSET_PATH=${TOOLSET_PATH}"
+export RETRIEVAL_TOOL_URL="http://${ip_address}:8889/v1/retrievaltool"
+export LANGCHAIN_API_KEY=${LANGCHAIN_API_KEY}
+export LANGCHAIN_TRACING_V2=${LANGCHAIN_TRACING_V2}
+export db_name=Chinook
+export db_path="sqlite:////home/user/chinook-db/Chinook_Sqlite.sqlite"
 export recursion_limit_worker=12
 export recursion_limit_supervisor=10
-export WORKER_AGENT_URL="http://${ip_address}:${AGENTQNA_WORKER_AGENT_SERVICE_PORT}/v1/chat/completions"
-export RETRIEVAL_TOOL_URL="http://${ip_address}:8889/v1/retrievaltool"
-export CRAG_SERVER=http://${ip_address}:18881
-
-export AGENTQNA_FRONTEND_PORT="9090"
-
-#retrieval_tool
+export CRAG_SERVER=http://${ip_address}:${CRAG_SERVER_PORT}
+export WORKER_AGENT_URL="http://${ip_address}:${WORKER_RAG_AGENT_PORT}/v1/chat/completions"
+export SQL_AGENT_URL="http://${ip_address}:${WORKER_SQL_AGENT_PORT}/v1/chat/completions"
+export HF_CACHE_DIR=${HF_CACHE_DIR}
+export HUGGINGFACEHUB_API_TOKEN=${HUGGINGFACEHUB_API_TOKEN}
+export no_proxy=${no_proxy}
+export http_proxy=${http_proxy}
+export https_proxy=${https_proxy}
+export EMBEDDING_MODEL_ID="BAAI/bge-base-en-v1.5"
+export RERANK_MODEL_ID="BAAI/bge-reranker-base"
 export TEI_EMBEDDING_ENDPOINT="http://${host_ip}:6006"
 export TEI_RERANKING_ENDPOINT="http://${host_ip}:8808"
-export REDIS_URL="redis://${host_ip}:26379"
+export REDIS_URL="redis://${host_ip}:6379"
 export INDEX_NAME="rag-redis"
+export RERANK_TYPE="tei"
 export MEGA_SERVICE_HOST_IP=${host_ip}
 export EMBEDDING_SERVICE_HOST_IP=${host_ip}
 export RETRIEVER_SERVICE_HOST_IP=${host_ip}
 export RERANK_SERVICE_HOST_IP=${host_ip}
 export BACKEND_SERVICE_ENDPOINT="http://${host_ip}:8889/v1/retrievaltool"
 export DATAPREP_SERVICE_ENDPOINT="http://${host_ip}:6007/v1/dataprep/ingest"
-export DATAPREP_GET_FILE_ENDPOINT="http://${host_ip}:6007/v1/dataprep/get"
-export DATAPREP_DELETE_FILE_ENDPOINT="http://${host_ip}:6007/v1/dataprep/delete"
+export DATAPREP_GET_FILE_ENDPOINT="http://${host_ip}:6008/v1/dataprep/get"
+export DATAPREP_DELETE_FILE_ENDPOINT="http://${host_ip}:6009/v1/dataprep/delete"
+
+echo ${WORKER_RAG_AGENT_PORT} > ${WORKPATH}/WORKER_RAG_AGENT_PORT_tmp
+echo ${WORKER_SQL_AGENT_PORT} > ${WORKPATH}/WORKER_SQL_AGENT_PORT_tmp
+echo ${SUPERVISOR_REACT_AGENT_PORT} > ${WORKPATH}/SUPERVISOR_REACT_AGENT_PORT_tmp
+echo ${CRAG_SERVER_PORT} > ${WORKPATH}/CRAG_SERVER_PORT_tmp
 
+echo "Downloading chinook data..."
+echo Y | rm -R chinook-database
+git clone https://github.com/lerocha/chinook-database.git
+echo Y | rm -R ../../../../../AgentQnA/tests/Chinook_Sqlite.sqlite
+cp chinook-database/ChinookDatabase/DataSources/Chinook_Sqlite.sqlite ../../../../../AgentQnA/tests
+
+docker compose -f ../../../../../DocIndexRetriever/docker_compose/intel/cpu/xeon/compose.yaml up -d
 docker compose -f compose.yaml up -d
+
+n=0
+until [[ "$n" -ge 100 ]]; do
+    docker logs tgi-service > ${WORKPATH}/tgi_service_start.log
+    if grep -q Connected ${WORKPATH}/tgi_service_start.log; then
+        break
+    fi
+    sleep 10s
+    n=$((n+1))
+done
+
+echo "Starting CRAG server"
+docker run -d --runtime=runc --name=kdd-cup-24-crag-service -p=${CRAG_SERVER_PORT}:8000 docker.io/aicrowd/kdd-cup-24-crag-mock-api:v0