This message was deleted.
# general
s
This message was deleted.
t
To resolve the issue of connecting Apache Druid to MinIO in a local Docker setup, ensure that you explicitly set a dummy region for the S3 client in Druid’s configuration, even if MinIO doesn’t require it, as Druid expects a region for S3 connections. Double-check that your MinIO instance is properly configured and accessible from Druid containers, ensuring they’re on the same Docker network for seamless communication. Also, verify that the S3 extensions are correctly loaded in Druid and that there’s no version compatibility issue with the Druid Docker image you’re using. Reviewing the logs of both Druid and MinIO can provide additional insights into the connection issue, and if necessary, network traffic analysis might help identify any underlying networking or endpoint configuration problems.
a
just set a random region. When I was testing it, I added this line in env file:
Copy code
AWS_REGION=us-east-1
I notice another thing in your compose file - I see that
druid.s3.endpoint.url
is set to localhost:9000. That's incorrect. You're asking your docker containers to find a minio server also running within the same container at port 9000. Set it to the right IP
example: Here's my .env file excerpt:
Copy code
druid_storage_type=s3
druid_storage_bucket=druid
druid_storage_baseKey=segments
druid_s3_enablePathStyleAccess=true
druid_s3_endpoint_signingRegion=us-east-1
druid_s3_endpoint_url=<http://minio-nginx:9000/>
Here's my compose file: (Pasting only relevant sections)
Copy code
x-minio-common: &minio-common
  image: <image>
  command: server --console-address ":9001" <http://minio>{1...4}/data{1...2}
  networks:
    - sonic-overlay
  env_file:
    # Since the container runs on node labeled sonic01, make sure the minio.env file is available there
    - minio.env
  healthcheck:
    test: ["CMD", "curl", "-f", "<http://localhost:9000/minio/health/live>"]
    interval: 30s
    timeout: 20s
    retries: 3
  deploy:
      placement:
        constraints:
        - node.labels.id==sonic01
...
...
services:
  # MinIO Services
  minio1:
    <<: *minio-common
    hostname: minio1
    volumes:
      - data1-1:/data1
      - data1-2:/data2

  minio2:
    <<: *minio-common
    hostname: minio2
    volumes:
      - data2-1:/data1
      - data2-2:/data2

  minio3:
    <<: *minio-common
    hostname: minio3
    volumes:
      - data3-1:/data1
      - data3-2:/data2

  minio4:
    <<: *minio-common
    hostname: minio4
    volumes:
      - data4-1:/data1
      - data4-2:/data2

  minio-nginx:
    image: <nginx>
    hostname: nginx
    deploy:
      placement:
        constraints:
        - node.labels.id==sonic01
    volumes:
      # Since the container runs on node labeled sonic01, make sure the minio_nginx.conf is available there
      - ./minio_nginx.conf:/etc/nginx/nginx.conf:ro
    networks:
      - sonic-overlay
    ports:
      - "9000:9000"
      - "9001:9001"
    depends_on:
      - minio1
      - minio2
      - minio3
      - minio4

volumes:
  druid-pg:

  # MinIO Volumes
  data1-1:
  data1-2:
  data2-1:
  data2-2:
  data3-1:
  data3-2:
  data4-1:
  data4-2:
d
Hello, thank you very much for your reply. It seems that I no longer have the problem that appeared to me, change the theme of the region and I put druid_s3_endpoint_url=http://nginx:9000/. Now I get the following error:
Copy code
Cannot construct instance of `org.apache.druid.data.input.s3.S3InputSourceConfig`, problem: accessKeyId cannot be null if secretAccessKey is given at [Source: (org.eclipse.jetty.server.HttpInputOverHTTP); line: 1, column: 157] (through reference chain: org.apache.druid. indexing.overlord.sampler.IndexTaskSamplerSpec["spec"]->org.apache.druid.indexing.common.task.IndexTask$IndexIngestionSpec["ioConfig"]->org.apache.druid.indexing.common.task.IndexTask$IndexIOConfig["inputSource"]->org.apache.druid.data.input.s3.S3InputSource["properties"])
This is my current environment file:
Copy code
# Java tuning
#DRUID_XMX=1g
#DRUID_XMS=1g
#DRUID_MAXNEWSIZE=250m
#DRUID_NEWSIZE=250m
#DRUID_MAXDIRECTMEMORYSIZE=6172m
DRUID_SINGLE_NODE_CONF=micro-quickstart

druid_emitter_logging_logLevel=debug

druid_extensions_loadList=["druid-s3-extensions", "druid-histogram", "druid-datasketches", "druid-lookups-cached-global", "postgresql-metadata-storage", "druid-multi-stage-query"]

druid_zk_service_host=zookeeper

druid_metadata_storage_host=
druid_metadata_storage_type=postgresql
druid_metadata_storage_connector_connectURI=jdbc:<postgresql://postgres:5432/druid>
druid_metadata_storage_connector_user=druid
druid_metadata_storage_connector_password=FoolishPassword

druid_coordinator_balancer_strategy=cachingCost

druid_indexer_runner_javaOptsArray=["-server", "-Xmx1g", "-Xms1g", "-XX:MaxDirectMemorySize=3g", "-Duser.timezone=UTC", "-Dfile.encoding=UTF-8", "-Djava.util.logging.manager=org.apache.logging.log4j.jul.LogManager"]
druid_indexer_fork_property_druid_processing_buffer_sizeBytes=256MiB

#druid_storage_type=local
#druid_storage_storageDirectory=/opt/shared/segments
#druid_indexer_logs_type=file
#druid_indexer_logs_directory=/opt/shared/indexing-logs

druid.s3.accessKey=iUpwaWMpOo09GYvnEH4W
druid.s3.secretKey=NjRoq0omCcOZH9el52qavWiGMltSLhdD8MOIeJNV
druid.s3.protocol=http
druid.s3.enablePathStyleAccess=true
druid.s3.endpoint.signingRegion=us-east-1
druid.s3.endpoint.url=<http://nginx:9000/>


druid.storage.type=s3
druid.storage.bucket=prueba
druid.storage.baseKey=segments

druid.indexer.logs.type=s3
druid.indexer.logs.s3Bucket=prueba
druid.indexer.logs.s3Prefix=indexing-logs
AWS_REGION=us-east-1


druid_processing_numThreads=2
druid_processing_numMergeBuffers=2

DRUID_LOG4J=<?xml version="1.0" encoding="UTF-8" ?><Configuration status="WARN"><Appenders><Console name="Console" target="SYSTEM_OUT"><PatternLayout pattern="%d{ISO8601} %p [%t] %c - %m%n"/></Console></Appenders><Loggers><Root level="info"><AppenderRef ref="Console"/></Root><Logger name="org.apache.druid.jetty.RequestLog" additivity="false" level="DEBUG"><AppenderRef ref="Console"/></Logger></Loggers></Configuration>
My docker-compose for minio:
Copy code
version: '3.7'

# Settings and configurations that are common for all containers
x-minio-common: &minio-common
  image: <http://quay.io/minio/minio:RELEASE.2023-06-19T19-52-50Z|quay.io/minio/minio:RELEASE.2023-06-19T19-52-50Z>
  restart: always
  command: server --console-address ":9001" <http://minio>{1...4}/data{1...2}
  expose:
    - "9000"
    - "9001"
  # environment:
    # MINIO_ROOT_USER: minioadmin
    # MINIO_ROOT_PASSWORD: minioadmin
  healthcheck:
    test: ["CMD", "curl", "-f", "<http://localhost:9000/minio/health/live>"]
    interval: 30s
    timeout: 20s
    retries: 3
  networks: 
    - apachedruid_default

# starts 4 docker containers running minio server instances.
# using nginx reverse proxy, load balancing, you can access
# it through port 9000.
services:
  minio1:
    <<: *minio-common
    hostname: minio1
    volumes:
      - data1-1:/data1
      - data1-2:/data2

    
  minio2:
    <<: *minio-common
    hostname: minio2
    volumes:
      - data2-1:/data1
      - data2-2:/data2
    

  minio3:
    <<: *minio-common
    hostname: minio3
    volumes:
      - data3-1:/data1
      - data3-2:/data2
    
  minio4:
    <<: *minio-common
    hostname: minio4
    volumes:
      - data4-1:/data1
      - data4-2:/data2
    

  nginx:
    image: nginx:1.19.2-alpine
    restart: always
    hostname: nginx
    volumes:
      - ./nginx.conf:/etc/nginx/nginx.conf:ro
    ports:
      - "9000:9000"
      - "9001:9001"
    depends_on:
      - minio1
      - minio2
      - minio3
      - minio4
    networks: 
    - apachedruid_default
   

## By default this config uses default local driver,
## For custom volumes replace with volume driver configuration.
volumes:
  data1-1:
  data1-2:
  data2-1:
  data2-2:
  data3-1:
  data3-2:
  data4-1:
  data4-2:

networks:
  apachedruid_default:
    external: true
my docker-compose for Apache Druid
Copy code
version: "2.2"

volumes:
  metadata_data: {}
  middle_var: {}
  historical_var: {}
  broker_var: {}
  coordinator_var: {}
  router_var: {}
  druid_shared: {}

services:
  postgres:
    container_name: postgres
    image: postgres:latest
    ports:
      - "5432:5432"
    volumes:
      - metadata_data:/var/lib/postgresql/data
    environment:
      - POSTGRES_PASSWORD=FoolishPassword
      - POSTGRES_USER=druid
      - POSTGRES_DB=druid
    restart: always  # Añadida la directiva restart: always

  # Need 3.5 or later for container nodes
  zookeeper:
    container_name: zookeeper
    image: zookeeper:3.5.10
    ports:
      - "2181:2181"
    environment:
      - ZOO_MY_ID=1
    restart: always  # Añadida la directiva restart: always

  coordinator:
    image: apache/druid:28.0.1
    container_name: coordinator
    volumes:
      - druid_shared:/opt/shared
      - coordinator_var:/opt/druid/var
    depends_on:
      - zookeeper
      - postgres
    ports:
      - "8081:8081"
    command:
      - coordinator
    env_file:
      - environment
    restart: always  # Añadida la directiva restart: always

  broker:
    image: apache/druid:28.0.1
    container_name: broker
    volumes:
      - broker_var:/opt/druid/var
    depends_on:
      - zookeeper
      - postgres
      - coordinator
    ports:
      - "8082:8082"
    command:
      - broker
    env_file:
      - environment
    restart: always  # Añadida la directiva restart: always

  historical:
    image: apache/druid:28.0.1
    container_name: historical
    volumes:
      - druid_shared:/opt/shared
      - historical_var:/opt/druid/var
    depends_on: 
      - zookeeper
      - postgres
      - coordinator
    ports:
      - "8083:8083"
    command:
      - historical
    env_file:
      - environment
    restart: always  # Añadida la directiva restart: always

  middlemanager:
    image: apache/druid:28.0.1
    container_name: middlemanager
    volumes:
      - druid_shared:/opt/shared
      - middle_var:/opt/druid/var
    depends_on: 
      - zookeeper
      - postgres
      - coordinator
    ports:
      - "8091:8091"
      - "8100-8105:8100-8105"
    command:
      - middleManager
    env_file:
      - environment
    restart: always  # Añadida la directiva restart: always

  router:
    image: apache/druid:28.0.1
    container_name: router
    volumes:
      - router_var:/opt/druid/var
    depends_on:
      - zookeeper
      - postgres
      - coordinator
    ports:
      - "8888:8888"
    command:
      - router
    env_file:
      - environment
    restart: always  # Añadida la directiva restart: always
This is the image that appears. I do not put neither secretkey nor acceskey because they would be annulled when being in the configuration file.
a
If somebody stumbles on this thread and is facing the same issue - the problem was in the .env file, all config values must be expressed as
druid_storage_type
. Not separated by dots, underscores must be used.