Chuyển đến nội dung chính

Lesson 22: FHIR Performance, Scalability and Monitoring

Optimize FHIR Server performance, caching strategies, database indexing, horizontal scaling, load balancing, CDN for static content, monitoring with Prometheus/Grafana, health checks, health SLA.

🏗️ Architecture — Lesson 22 Lesson 22: FHIR Performance, Scalability and Monitoring

HL7 FHIR - Basic to Advanced Healthcare Data Standard

Part 7: Production, Scale and Future

xdev.asia

1. Performance Challenges in Healthcare

The medical FHIR Server system must meet high requirements: uptime ≥ 99.9%, response time < 500ms, millions of resources.

FactorChallengeTarget
Response timeComplex search on large datasets< 500ms (P95)
ThroughputMany hospitals query at the same time≥ 1000 req/s
Data volumesMillions of Patients, billions of ObservationsTB scale
Availability24/7 for emergency care≥ 99.9%
ConsistencyMedical data must be accurateStrong consistency

2. Database Optimization

PostgreSQL Indexing for HAPI FHIR

-- Index cho search by identifier (CCCD, BHYT)
CREATE INDEX idx_spidx_token_hash 
ON hfj_spidx_token (hash_identity, sp_value_normalized)
WHERE sp_missing = false;

-- Index cho search by name
CREATE INDEX idx_spidx_string_normalized 
ON hfj_spidx_string (hash_identity, sp_value_normalized)
WHERE sp_missing = false;

-- Index cho search by date (Observation.date)
CREATE INDEX idx_spidx_date_range 
ON hfj_spidx_date (hash_identity, sp_value_low, sp_value_high)
WHERE sp_missing = false;

-- Index cho _lastUpdated sorting
CREATE INDEX idx_res_updated 
ON hfj_resource (res_updated DESC, res_id)
WHERE res_deleted_at IS NULL;

-- Partitioning cho Observation (table lớn nhất)
CREATE TABLE hfj_res_observation PARTITION OF hfj_resource
FOR VALUES IN ('Observation')
PARTITION BY RANGE (res_updated);

CREATE TABLE hfj_res_observation_2025_q1 
PARTITION OF hfj_res_observation
FOR VALUES FROM ('2025-01-01') TO ('2025-04-01');

Query Optimization

# application.yml - HAPI FHIR tuning
hapi:
  fhir:
    # Search caching
    reuse_cached_search_results_millis: 60000
    
    # Pagination
    default_page_size: 20
    max_page_size: 200
    
    # Prefetch
    search_prefetch_thresholds:
      - 13  # offset 0
      - 50  # offset 1
      - 200 # offset 2+
    
    # Inline resource storage (tránh JOIN)
    inline_resource_storage_below_size: 4096

spring:
  datasource:
    hikari:
      maximum-pool-size: 50
      minimum-idle: 10
      connection-timeout: 5000
      idle-timeout: 300000

3. Caching Strategies

┌────────┐   ┌───────┐   ┌──────────┐   ┌──────┐
│ Client │──▶│ CDN/  │──▶│  Redis   │──▶│ FHIR │
│        │   │ Nginx │   │  Cache   │   │Server│
└────────┘   └───────┘   └──────────┘   └──────┘
// Redis cache cho CapabilityStatement và ValueSet
@Configuration
public class FhirCacheConfig {
    
    @Bean
    public CacheManager cacheManager(
            RedisConnectionFactory redisFactory) {
        RedisCacheConfiguration config = 
            RedisCacheConfiguration.defaultCacheConfig()
                .entryTtl(Duration.ofMinutes(30))
                .serializeValuesWith(
                    SerializationPair.fromSerializer(
                        new GenericJackson2JsonRedisSerializer()));
        
        return RedisCacheManager.builder(redisFactory)
            .cacheDefaults(config)
            .withCacheConfiguration("metadata",
                config.entryTtl(Duration.ofHours(1)))
            .withCacheConfiguration("valueset",
                config.entryTtl(Duration.ofHours(24)))
            .build();
    }
}

// Interceptor: cache CapabilityStatement
@Component
public class MetadataCacheInterceptor {
    
    @Autowired
    private CacheManager cacheManager;
    
    @Hook(Pointcut.SERVER_CAPABILITY_STATEMENT_GENERATED)
    public void cacheCapabilityStatement(
            IBaseConformance cs) {
        cacheManager.getCache("metadata")
            .put("capability", cs);
    }
}

HTTP Caching Headers

# nginx.conf
location /fhir/metadata {
    proxy_pass http://fhir-server:8080;
    proxy_cache fhir_cache;
    proxy_cache_valid 200 1h;
    add_header X-Cache-Status $upstream_cache_status;
}

location /fhir/ValueSet {
    proxy_pass http://fhir-server:8080;
    proxy_cache fhir_cache;
    proxy_cache_valid 200 24h;
    proxy_cache_key "$request_uri";
}

# Dynamic resources — no cache
location /fhir/Patient {
    proxy_pass http://fhir-server:8080;
    add_header Cache-Control "no-store";
}

4. Horizontal Scaling

# docker-compose-ha.yml
services:
  nginx:
    image: nginx:alpine
    ports:
      - "443:443"
    volumes:
      - ./nginx.conf:/etc/nginx/nginx.conf
    depends_on:
      - fhir-1
      - fhir-2
      - fhir-3

  fhir-1:
    image: hapiproject/hapi:latest
    environment:
      - SPRING_DATASOURCE_URL=jdbc:postgresql://pgpool:5432/hapi_fhir
      - JAVA_OPTS=-Xmx2g
  
  fhir-2:
    image: hapiproject/hapi:latest
    environment:
      - SPRING_DATASOURCE_URL=jdbc:postgresql://pgpool:5432/hapi_fhir
      - JAVA_OPTS=-Xmx2g
  
  fhir-3:
    image: hapiproject/hapi:latest
    environment:
      - SPRING_DATASOURCE_URL=jdbc:postgresql://pgpool:5432/hapi_fhir
      - JAVA_OPTS=-Xmx2g

  pgpool:
    image: bitnami/pgpool:latest
    environment:
      - PGPOOL_BACKEND_NODES=0:pg-primary:5432,1:pg-replica:5432
      - PGPOOL_ENABLE_LOAD_BALANCING=yes
      - PGPOOL_SR_CHECK_USER=repmgr

  pg-primary:
    image: bitnami/postgresql-repmgr:16
    environment:
      - POSTGRESQL_DATABASE=hapi_fhir
      - REPMGR_NODE_TYPE=primary

  pg-replica:
    image: bitnami/postgresql-repmgr:16
    environment:
      - REPMGR_NODE_TYPE=standby
      - REPMGR_PRIMARY_HOST=pg-primary

  redis:
    image: redis:7-alpine
    command: redis-server --maxmemory 1gb --maxmemory-policy allkeys-lru

5. Monitoring with Prometheus + Grafana

Expose Metrics

# application.yml
management:
  endpoints:
    web:
      exposure:
        include: health,info,prometheus
  metrics:
    export:
      prometheus:
        enabled: true
    tags:
      application: fhir-server

Prometheus Config

# prometheus.yml
scrape_configs:
  - job_name: 'fhir-server'
    metrics_path: '/actuator/prometheus'
    scrape_interval: 15s
    static_configs:
      - targets:
          - 'fhir-1:8080'
          - 'fhir-2:8080'
          - 'fhir-3:8080'

Key Metrics need tracking

MetricDescriptionAlert threshold
http_server_requests_seconds_p95P95 response time> 500ms
http_server_requests_totalRequest rateAnomaly detection
jvm_memory_used_bytesJVM heap usage> 80%
hikaricp_connections_activeActive DB connections> 80% pools
fhir_search_duration_secondsSearch query time> 1s
fhir_resource_countTotal resourcesGrowth rate
pg_stat_activity_countPostgreSQL connections> 90% max
disk_usage_percentDisk space> 85%

Health Check Endpoint

@Component
public class FhirHealthIndicator implements HealthIndicator {
    
    @Autowired
    private IGenericClient fhirClient;
    
    @Override
    public Health health() {
        try {
            CapabilityStatement cs = fhirClient
                .capabilities()
                .ofType(CapabilityStatement.class)
                .execute();
            
            return Health.up()
                .withDetail("fhirVersion", cs.getFhirVersion().toCode())
                .withDetail("resourceCount", 
                    cs.getRest().get(0).getResource().size())
                .build();
        } catch (Exception e) {
            return Health.down()
                .withException(e)
                .build();
        }
    }
}

6. Alerting Rules

# alerting-rules.yml
groups:
  - name: fhir-server
    rules:
      - alert: FHIRHighResponseTime
        expr: |
          histogram_quantile(0.95,
            rate(http_server_requests_seconds_bucket{uri=~"/fhir/.*"}[5m])
          ) > 0.5
        for: 5m
        labels:
          severity: warning
        annotations:
          summary: "FHIR Server P95 response time > 500ms"

      - alert: FHIRServerDown
        expr: up{job="fhir-server"} == 0
        for: 1m
        labels:
          severity: critical
        annotations:
          summary: "FHIR Server instance down"

      - alert: FHIRDatabaseConnectionExhausted
        expr: |
          hikaricp_connections_active / 
          hikaricp_connections_max > 0.8
        for: 5m
        labels:
          severity: warning

7. Summary

  • Database optimization — Indexing, partitioning, connection pooling for PostgreSQL

  • Caching — Redis for metadata/ValueSet, Nginx reverse proxy cache

  • Horizontal scaling — Multiple FHIR instances + load balancer + PgPool

  • Monitoring — Prometheus metrics, Grafana dashboards, alerting rules

  • Health checks — Custom HealthIndicator, readiness/liveness probes

  • Medical SLAs — Uptime ≥ 99.9%, P95 < 500ms, audit trail 7+ years