* update readme gaudi part & add tei-gaudi params Signed-off-by: letonghan <letong.han@intel.com> * modify supported habana driver version Signed-off-by: letonghan <letong.han@intel.com> * update env set part Signed-off-by: letonghan <letong.han@intel.com> * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * add example for no_proxy Signed-off-by: letonghan <letong.han@intel.com> * add an example of public ip Signed-off-by: letonghan <letong.han@intel.com> --------- Signed-off-by: letonghan <letong.han@intel.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
164 lines
4.8 KiB
YAML
164 lines
4.8 KiB
YAML
|
|
# Copyright (C) 2024 Intel Corporation
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
|
|
version: "3.8"
|
|
|
|
services:
|
|
tei-embedding-service:
|
|
image: opea/tei-gaudi:latest
|
|
container_name: tei-embedding-gaudi-server
|
|
ports:
|
|
- "3001:80"
|
|
volumes:
|
|
- "./data:/data"
|
|
runtime: habana
|
|
cap_add:
|
|
- SYS_NICE
|
|
ipc: host
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
HABANA_VISIBLE_DEVICES: all
|
|
OMPI_MCA_btl_vader_single_copy_mechanism: none
|
|
MAX_WARMUP_SEQUENCE_LENGTH: 512
|
|
INIT_HCCL_ON_ACQUIRE: 0
|
|
ENABLE_EXPERIMENTAL_FLAGS: true
|
|
command: --model-id ${EMBEDDING_MODEL_ID} --auto-truncate
|
|
embedding:
|
|
image: opea/embedding-tei:latest
|
|
container_name: embedding-tei-server
|
|
depends_on:
|
|
- tei-embedding-service
|
|
ports:
|
|
- "3002:6000"
|
|
ipc: host
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
TEI_EMBEDDING_ENDPOINT: ${TEI_EMBEDDING_ENDPOINT}
|
|
LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
|
|
LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
|
|
LANGCHAIN_PROJECT: "opea-embedding-service"
|
|
restart: unless-stopped
|
|
web-retriever:
|
|
image: opea/web-retriever-chroma:latest
|
|
container_name: web-retriever-chroma-server
|
|
ports:
|
|
- "3003:7077"
|
|
ipc: host
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
TEI_EMBEDDING_ENDPOINT: ${TEI_EMBEDDING_ENDPOINT}
|
|
GOOGLE_API_KEY: ${GOOGLE_API_KEY}
|
|
GOOGLE_CSE_ID: ${GOOGLE_CSE_ID}
|
|
restart: unless-stopped
|
|
tei-reranking-service:
|
|
image: ghcr.io/huggingface/text-embeddings-inference:cpu-1.2
|
|
container_name: tei-reranking-server
|
|
ports:
|
|
- "3004:80"
|
|
volumes:
|
|
- "./data:/data"
|
|
shm_size: 1g
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
command: --model-id ${RERANK_MODEL_ID} --auto-truncate
|
|
reranking:
|
|
image: opea/reranking-tei:latest
|
|
container_name: reranking-tei-xeon-server
|
|
depends_on:
|
|
- tei-reranking-service
|
|
ports:
|
|
- "3005:8000"
|
|
ipc: host
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
TEI_RERANKING_ENDPOINT: ${TEI_RERANKING_ENDPOINT}
|
|
HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
|
|
LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
|
|
LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
|
|
LANGCHAIN_PROJECT: "opea-reranking-service"
|
|
restart: unless-stopped
|
|
tgi-service:
|
|
image: ghcr.io/huggingface/tgi-gaudi:2.0.1
|
|
container_name: tgi-gaudi-server
|
|
ports:
|
|
- "3006:80"
|
|
volumes:
|
|
- "./data:/data"
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
HF_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
|
|
HF_HUB_DISABLE_PROGRESS_BARS: 1
|
|
HF_HUB_ENABLE_HF_TRANSFER: 0
|
|
HABANA_VISIBLE_DEVICES: all
|
|
OMPI_MCA_btl_vader_single_copy_mechanism: none
|
|
runtime: habana
|
|
cap_add:
|
|
- SYS_NICE
|
|
ipc: host
|
|
command: --model-id ${LLM_MODEL_ID} --max-input-length 1024 --max-total-tokens 2048
|
|
llm:
|
|
image: opea/llm-tgi:latest
|
|
container_name: llm-tgi-gaudi-server
|
|
depends_on:
|
|
- tgi-service
|
|
ports:
|
|
- "3007:9000"
|
|
ipc: host
|
|
environment:
|
|
no_proxy: ${no_proxy}
|
|
http_proxy: ${http_proxy}
|
|
https_proxy: ${https_proxy}
|
|
TGI_LLM_ENDPOINT: ${TGI_LLM_ENDPOINT}
|
|
HUGGINGFACEHUB_API_TOKEN: ${HUGGINGFACEHUB_API_TOKEN}
|
|
HF_HUB_DISABLE_PROGRESS_BARS: 1
|
|
HF_HUB_ENABLE_HF_TRANSFER: 0
|
|
LANGCHAIN_API_KEY: ${LANGCHAIN_API_KEY}
|
|
LANGCHAIN_TRACING_V2: ${LANGCHAIN_TRACING_V2}
|
|
LANGCHAIN_PROJECT: "opea-llm-service"
|
|
restart: unless-stopped
|
|
searchqna-gaudi-backend-server:
|
|
image: opea/searchqna:latest
|
|
container_name: searchqna-gaudi-backend-server
|
|
depends_on:
|
|
- tei-embedding-service
|
|
- embedding
|
|
- web-retriever
|
|
- tei-reranking-service
|
|
- reranking
|
|
- tgi-service
|
|
- llm
|
|
ports:
|
|
- "3008:8888"
|
|
environment:
|
|
- no_proxy=${no_proxy}
|
|
- https_proxy=${https_proxy}
|
|
- http_proxy=${http_proxy}
|
|
- MEGA_SERVICE_HOST_IP=${MEGA_SERVICE_HOST_IP}
|
|
- EMBEDDING_SERVICE_HOST_IP=${EMBEDDING_SERVICE_HOST_IP}
|
|
- WEB_RETRIEVER_SERVICE_HOST_IP=${WEB_RETRIEVER_SERVICE_HOST_IP}
|
|
- RERANK_SERVICE_HOST_IP=${RERANK_SERVICE_HOST_IP}
|
|
- LLM_SERVICE_HOST_IP=${LLM_SERVICE_HOST_IP}
|
|
- EMBEDDING_SERVICE_PORT=${EMBEDDING_SERVICE_PORT}
|
|
- WEB_RETRIEVER_SERVICE_PORT=${WEB_RETRIEVER_SERVICE_PORT}
|
|
- RERANK_SERVICE_PORT=${RERANK_SERVICE_PORT}
|
|
- LLM_SERVICE_PORT=${LLM_SERVICE_PORT}
|
|
ipc: host
|
|
restart: always
|
|
|
|
networks:
|
|
default:
|
|
driver: bridge
|