#!/usr/bin/env bash
set -Eeuo pipefail

# ============================================================
# NextLimit - Build automatique des datasets Europe
#
# Dépendance hôte :
#   Docker uniquement
#
# Le script :
#   1. construit l'image Docker NextLimit
#   2. récupère l'index officiel Geofabrik
#   3. détecte automatiquement les pays européens
#   4. télécharge chaque PBF
#   5. extrait uniquement le réseau automobile
#   6. supprime les tags inutiles
#   7. valide le PBF
#   8. génère le SHA256
#   9. supprime les gros fichiers intermédiaires
#  10. génère un catalogue JSON final
#
# Usage :
#
#   ./europe.sh
#
# Reprendre après interruption :
#
#   ./europe.sh
#
# Le script ignore automatiquement les datasets déjà générés.
#
# Reconstruire un pays :
#
#   FORCE=1 ./europe.sh
#
# Conserver les PBF sources :
#
#   KEEP_SOURCE=1 ./europe.sh
#
# Conserver les PBF intermédiaires :
#
#   KEEP_INTERMEDIATE=1 ./europe.sh
#
# Tester sur quelques pays :
#
#   COUNTRIES="FR,BE,LU,DE" ./europe.sh
#
# ============================================================


SCRIPT_DIR="$(
    cd "$(dirname "${BASH_SOURCE[0]}")"
    pwd
)"

IMAGE_NAME="${IMAGE_NAME:-nextlimit-osm-tool:latest}"

DOCKERFILE="${SCRIPT_DIR}/Dockerfile"

DATA_DIR="${DATA_DIR:-${SCRIPT_DIR}/data}"

KEEP_SOURCE="${KEEP_SOURCE:-0}"
KEEP_INTERMEDIATE="${KEEP_INTERMEDIATE:-0}"
FORCE="${FORCE:-0}"

# Optionnel :
# COUNTRIES="FR,BE,DE"
COUNTRIES="${COUNTRIES:-}"

INDEX_URL="https://download.geofabrik.de/index-v1-nogeom.json"


# ============================================================
# Helpers
# ============================================================

log() {
    printf '\n\033[1;34m[NextLimit]\033[0m %s\n' "$*"
}

success() {
    printf '\n\033[1;32m[NextLimit]\033[0m %s\n' "$*"
}

warning() {
    printf '\n\033[1;33m[NextLimit]\033[0m %s\n' "$*"
}

die() {
    printf '\n\033[1;31m[ERREUR]\033[0m %s\n' "$*" >&2
    exit 1
}


# ============================================================
# Docker
# ============================================================

command -v docker >/dev/null 2>&1 \
    || die "Docker n'est pas installé."

docker info >/dev/null 2>&1 \
    || die "Docker n'est pas démarré ou inaccessible."

[[ -f "${DOCKERFILE}" ]] \
    || die "Dockerfile introuvable : ${DOCKERFILE}"

mkdir -p "${DATA_DIR}"


# ============================================================
# Construction image
# ============================================================

log "Construction / vérification de l'image Docker..."

docker build \
    --tag "${IMAGE_NAME}" \
    - < "${DOCKERFILE}"


# ============================================================
# Traitement Europe
# ============================================================

log "Lancement de la génération Europe..."

docker run \
    --rm \
    --interactive \
    --user "$(id -u):$(id -g)" \
    --env "INDEX_URL=${INDEX_URL}" \
    --env "KEEP_SOURCE=${KEEP_SOURCE}" \
    --env "KEEP_INTERMEDIATE=${KEEP_INTERMEDIATE}" \
    --env "FORCE=${FORCE}" \
    --env "COUNTRIES=${COUNTRIES}" \
    --volume "${DATA_DIR}:/work/data" \
    "${IMAGE_NAME}" \
    bash -s <<'CONTAINER_SCRIPT'

set -Eeuo pipefail


# ============================================================
# Configuration
# ============================================================

ROOT="/work/data"

SOURCES_DIR="${ROOT}/sources"
OUTPUT_DIR="${ROOT}/packages"
TMP_DIR="${ROOT}/tmp"
REPORT_DIR="${ROOT}/reports"

INDEX_FILE="${ROOT}/geofabrik-index.json"
COUNTRY_LIST="${ROOT}/europe-countries.tsv"

CATALOG_JSON="${ROOT}/catalog.json"
SUMMARY_CSV="${ROOT}/build-summary.csv"
FAILED_FILE="${ROOT}/failed.txt"

mkdir -p \
    "${SOURCES_DIR}" \
    "${OUTPUT_DIR}" \
    "${TMP_DIR}" \
    "${REPORT_DIR}"

: > "${FAILED_FILE}"


# ============================================================
# Helpers
# ============================================================

log() {
    printf '\n\033[1;34m[Docker / NextLimit]\033[0m %s\n' "$*"
}

success() {
    printf '\n\033[1;32m[Docker / NextLimit]\033[0m %s\n' "$*"
}

warning() {
    printf '\n\033[1;33m[Docker / NextLimit]\033[0m %s\n' "$*"
}

human_size() {
    numfmt \
        --to=iec-i \
        --suffix=B \
        "$1" 2>/dev/null \
        || echo "${1} bytes"
}


# ============================================================
# Versions
# ============================================================

log "Environnement"

python3 --version
osmium --version

python3 - <<'PY'
import osmium
print("PyOsmium : OK")
PY


# ============================================================
# Récupération index Geofabrik
# ============================================================

log "Téléchargement de l'index Geofabrik..."

curl \
    --fail \
    --location \
    --retry 5 \
    --retry-delay 5 \
    --output "${INDEX_FILE}" \
    "${INDEX_URL}"


# ============================================================
# Génération liste pays Europe
#
# L'index Geofabrik fournit :
#
# id
# parent
# iso3166-1:alpha2
# urls.pbf
#
# On conserve les extractions :
#
# - appartenant à Europe
# - disposant d'un ISO 3166-1
# - disposant d'un PBF
#
# ISO3166-1 est attaché par Geofabrik à la plus petite
# extraction couvrant réellement le pays.
# ============================================================

log "Détection automatique des pays européens..."

python3 - \
    "${INDEX_FILE}" \
    "${COUNTRY_LIST}" \
    "${COUNTRIES}" <<'PY'

import json
import sys


index_file = sys.argv[1]
output_file = sys.argv[2]
requested_raw = sys.argv[3].strip()


requested = {
    x.strip().upper()
    for x in requested_raw.split(",")
    if x.strip()
}


with open(index_file, "r", encoding="utf-8") as f:
    data = json.load(f)


features = data.get("features", [])


# ------------------------------------------------------------
# Index par id
# ------------------------------------------------------------

by_id = {}

for feature in features:

    props = feature.get("properties", {})

    fid = props.get("id")

    if fid:
        by_id[fid] = props


# ------------------------------------------------------------
# Vérifie récursivement qu'une extraction appartient à Europe
# ------------------------------------------------------------

def is_in_europe(props):

    fid = props.get("id", "")

    if fid == "europe":
        return True

    if fid.startswith("europe/"):
        return True

    parent = props.get("parent")

    visited = set()

    while parent and parent not in visited:

        visited.add(parent)

        if parent == "europe":
            return True

        current = by_id.get(parent)

        if not current:
            break

        parent = current.get("parent")

    return False


rows = []


for props in by_id.values():

    if not is_in_europe(props):
        continue

    iso_codes = props.get("iso3166-1:alpha2") or []

    if not iso_codes:
        continue

    urls = props.get("urls") or {}

    pbf = urls.get("pbf")

    if not pbf:
        continue

    name = props.get("name") or props.get("id")

    region_id = props.get("id")

    iso_codes = [
        str(x).upper()
        for x in iso_codes
    ]

    # Si filtrage demandé :
    if requested:

        if not requested.intersection(iso_codes):
            continue

    iso = ",".join(iso_codes)

    rows.append(
        (
            iso,
            name,
            region_id,
            pbf,
        )
    )


rows.sort(
    key=lambda row: (
        row[0],
        row[1],
    )
)


with open(output_file, "w", encoding="utf-8") as f:

    for row in rows:

        f.write(
            "\t".join(row)
            + "\n"
        )


print(
    f"{len(rows)} extraction(s) européenne(s) détectée(s)."
)

for iso, name, region_id, pbf in rows:
    print(f"  {iso:8s} {name}")

PY


# ============================================================
# Filtre Osmium
# ============================================================

FILTER_FILE="${TMP_DIR}/nextlimit-filter.txt"

cat > "${FILTER_FILE}" <<'EOF'

w/highway=motorway
w/highway=motorway_link

w/highway=trunk
w/highway=trunk_link

w/highway=primary
w/highway=primary_link

w/highway=secondary
w/highway=secondary_link

w/highway=tertiary
w/highway=tertiary_link

w/highway=unclassified
w/highway=residential
w/highway=living_street
w/highway=service
w/highway=road

r/type=restriction
r/type=restriction:*

EOF


# ============================================================
# Script PyOsmium de nettoyage
# ============================================================

STRIP_SCRIPT="${TMP_DIR}/strip-tags.py"

cat > "${STRIP_SCRIPT}" <<'PY'

#!/usr/bin/env python3

import os
import sys
import osmium


INPUT = sys.argv[1]
OUTPUT = sys.argv[2]


WAY_TAGS = {

    "highway",

    "maxspeed",
    "maxspeed:forward",
    "maxspeed:backward",
    "maxspeed:conditional",
    "maxspeed:type",
    "source:maxspeed",
    "zone:maxspeed",

    "oneway",
    "junction",

    "access",
    "vehicle",
    "motor_vehicle",
    "motorcar",

    "service",

    # Conservés pour le map matching / debug.
    # Pourront être supprimés ultérieurement si inutiles.
    "ref",
    "name",
}


WAY_PREFIXES = (
    "maxspeed:",
)


RELATION_TAGS = {

    "type",

    "restriction",

    "restriction:conditional",

    "except",
}


RELATION_PREFIXES = (
    "restriction:",
)


def filter_tags(tags, allowed, prefixes=()):

    result = {}

    for tag in tags:

        key = tag.k

        if key in allowed:

            result[key] = tag.v

            continue

        if any(
            key.startswith(prefix)
            for prefix in prefixes
        ):

            result[key] = tag.v

    return result


class Handler(osmium.SimpleHandler):

    def __init__(self, writer):

        super().__init__()

        self.writer = writer

        self.nodes = 0
        self.ways = 0
        self.relations = 0


    def node(self, node):

        self.nodes += 1

        if len(node.tags):

            node = node.replace(tags={})

        self.writer.add_node(node)


    def way(self, way):

        self.ways += 1

        tags = filter_tags(
            way.tags,
            WAY_TAGS,
            WAY_PREFIXES,
        )

        self.writer.add_way(
            way.replace(tags=tags)
        )


    def relation(self, relation):

        self.relations += 1

        tags = filter_tags(
            relation.tags,
            RELATION_TAGS,
            RELATION_PREFIXES,
        )

        self.writer.add_relation(
            relation.replace(tags=tags)
        )


if os.path.exists(OUTPUT):
    os.unlink(OUTPUT)


writer = osmium.SimpleWriter(OUTPUT)

handler = Handler(writer)


try:

    handler.apply_file(
        INPUT,
        locations=False,
    )

finally:

    writer.close()


print(
    f"nodes={handler.nodes} "
    f"ways={handler.ways} "
    f"relations={handler.relations}"
)

PY


# ============================================================
# Summary CSV
# ============================================================

echo \
"iso,name,region_id,source_url,source_bytes,filtered_bytes,final_bytes,reduction_percent,status" \
> "${SUMMARY_CSV}"


# ============================================================
# Compteurs
# ============================================================

TOTAL=0
BUILT=0
SKIPPED=0
FAILED=0


# ============================================================
# Traitement pays par pays
# ============================================================

while IFS=$'\t' read -r \
    ISO \
    NAME \
    REGION_ID \
    PBF_URL

do

    [[ -z "${ISO}" ]] && continue

    TOTAL=$((TOTAL + 1))


    # --------------------------------------------------------
    # Identifiant fichier
    #
    # Quand plusieurs codes ISO couvrent une extraction :
    #
    # GG,JE
    #
    # devient :
    #
    # GG-JE
    # --------------------------------------------------------

    PACKAGE_ID="$(
        echo "${ISO}" \
        | tr ',' '-' \
        | tr '[:lower:]' '[:upper:]'
    )"


    SAFE_ID="$(
        echo "${REGION_ID}" \
        | sed 's#[^A-Za-z0-9._-]#-#g'
    )"


    SOURCE_FILE="${SOURCES_DIR}/${PACKAGE_ID}-latest.osm.pbf"

    FILTERED_FILE="${TMP_DIR}/${PACKAGE_ID}-filtered.osm.pbf"

    FINAL_FILE="${OUTPUT_DIR}/${PACKAGE_ID}.osm.pbf"

    SHA_FILE="${FINAL_FILE}.sha256"

    META_FILE="${OUTPUT_DIR}/${PACKAGE_ID}.json"


    echo
    echo "============================================================"
    echo "${PACKAGE_ID} - ${NAME}"
    echo "============================================================"


    # --------------------------------------------------------
    # Skip si déjà généré
    # --------------------------------------------------------

    if [[ "${FORCE:-0}" != "1" ]] \
        && [[ -s "${FINAL_FILE}" ]] \
        && [[ -s "${SHA_FILE}" ]] \
        && sha256sum -c "${SHA_FILE}" >/dev/null 2>&1

    then

        success "${PACKAGE_ID} déjà généré et valide."

        SKIPPED=$((SKIPPED + 1))

        continue

    fi


    # --------------------------------------------------------
    # Isolation des erreurs
    #
    # Un pays en échec ne bloque pas toute l'Europe.
    # --------------------------------------------------------

    set +e

    (

        set -Eeuo pipefail


        # ----------------------------------------------------
        # Download
        # ----------------------------------------------------

        log "${PACKAGE_ID} : téléchargement..."

        curl \
            --fail \
            --location \
            --retry 10 \
            --retry-delay 5 \
            --retry-all-errors \
            --continue-at - \
            --output "${SOURCE_FILE}" \
            "${PBF_URL}"


        osmium fileinfo "${SOURCE_FILE}" >/dev/null


        SOURCE_BYTES="$(
            stat -c%s "${SOURCE_FILE}"
        )"


        log "${PACKAGE_ID} : source $(human_size "${SOURCE_BYTES}")"


        # ----------------------------------------------------
        # Extraction réseau routier
        # ----------------------------------------------------

        log "${PACKAGE_ID} : extraction routière..."

        rm -f "${FILTERED_FILE}"


        osmium tags-filter \
            --expressions="${FILTER_FILE}" \
            --remove-tags \
            --overwrite \
            --output="${FILTERED_FILE}" \
            "${SOURCE_FILE}"


        FILTERED_BYTES="$(
            stat -c%s "${FILTERED_FILE}"
        )"


        # ----------------------------------------------------
        # Strip tags
        # ----------------------------------------------------

        log "${PACKAGE_ID} : optimisation NextLimit..."

        rm -f "${FINAL_FILE}"


        python3 \
            "${STRIP_SCRIPT}" \
            "${FILTERED_FILE}" \
            "${FINAL_FILE}"


        # ----------------------------------------------------
        # Validation
        # ----------------------------------------------------

        log "${PACKAGE_ID} : validation..."

        osmium fileinfo "${FINAL_FILE}" >/dev/null

        osmium check-refs "${FINAL_FILE}"


        FINAL_BYTES="$(
            stat -c%s "${FINAL_FILE}"
        )"


        REDUCTION="$(
            awk \
                -v source="${SOURCE_BYTES}" \
                -v final="${FINAL_BYTES}" \
                'BEGIN {
                    if (source == 0) {
                        print "0.00"
                    } else {
                        printf "%.2f", (1 - final/source) * 100
                    }
                }'
        )"


        # ----------------------------------------------------
        # Checksum
        #
        # Le chemin du sha doit être portable.
        # ----------------------------------------------------

        (
            cd "${OUTPUT_DIR}"

            sha256sum \
                "$(basename "${FINAL_FILE}")" \
                > "$(basename "${SHA_FILE}")"
        )


        # ----------------------------------------------------
        # Metadata package
        # ----------------------------------------------------

        python3 - \
            "${META_FILE}" \
            "${PACKAGE_ID}" \
            "${NAME}" \
            "${ISO}" \
            "${REGION_ID}" \
            "${PBF_URL}" \
            "${SOURCE_BYTES}" \
            "${FINAL_BYTES}" \
            "${REDUCTION}" <<'PY'

import datetime
import json
import sys


(
    output,
    package_id,
    name,
    iso,
    region_id,
    source_url,
    source_bytes,
    final_bytes,
    reduction,
) = sys.argv[1:]


data = {

    "id": package_id,

    "name": name,

    "iso3166_1": iso.split(","),

    "geofabrikRegionId": region_id,

    "source": source_url,

    "file": f"{package_id}.osm.pbf",

    "checksumFile": f"{package_id}.osm.pbf.sha256",

    "sourceBytes": int(source_bytes),

    "size": int(final_bytes),

    "reductionPercent": float(reduction),

    "generatedAt": (
        datetime.datetime.now(
            datetime.timezone.utc
        )
        .replace(microsecond=0)
        .isoformat()
    ),
}


with open(
    output,
    "w",
    encoding="utf-8",
) as f:

    json.dump(
        data,
        f,
        ensure_ascii=False,
        indent=2,
    )

PY


        # ----------------------------------------------------
        # Summary
        # ----------------------------------------------------

        printf \
            '"%s","%s","%s","%s",%s,%s,%s,%s,"OK"\n' \
            "${PACKAGE_ID}" \
            "${NAME//\"/\"\"}" \
            "${REGION_ID}" \
            "${PBF_URL}" \
            "${SOURCE_BYTES}" \
            "${FILTERED_BYTES}" \
            "${FINAL_BYTES}" \
            "${REDUCTION}" \
            >> "${SUMMARY_CSV}"


        echo
        echo "Source  : $(human_size "${SOURCE_BYTES}")"
        echo "Final   : $(human_size "${FINAL_BYTES}")"
        echo "Gain    : ${REDUCTION} %"


        # ----------------------------------------------------
        # Cleanup
        # ----------------------------------------------------

        if [[ "${KEEP_INTERMEDIATE:-0}" != "1" ]]; then

            rm -f "${FILTERED_FILE}"

        fi


        if [[ "${KEEP_SOURCE:-0}" != "1" ]]; then

            rm -f "${SOURCE_FILE}"

        fi


        success "${PACKAGE_ID} terminé."

    )

    STATUS=$?

    set -e


    if [[ "${STATUS}" -eq 0 ]]; then

        BUILT=$((BUILT + 1))

    else

        FAILED=$((FAILED + 1))

        warning "${PACKAGE_ID} en échec."

        echo \
            "${PACKAGE_ID}	${NAME}	${PBF_URL}" \
            >> "${FAILED_FILE}"

        rm -f "${FILTERED_FILE}"

    fi


done < "${COUNTRY_LIST}"


# ============================================================
# Génération catalogue global
# ============================================================

log "Génération du catalogue Europe..."

python3 - \
    "${OUTPUT_DIR}" \
    "${CATALOG_JSON}" <<'PY'

import datetime
import json
import pathlib
import sys


package_dir = pathlib.Path(sys.argv[1])
output = pathlib.Path(sys.argv[2])


regions = []


for filename in sorted(
    package_dir.glob("*.json")
):

    try:

        with filename.open(
            "r",
            encoding="utf-8",
        ) as f:

            regions.append(
                json.load(f)
            )

    except Exception as exc:

        print(
            f"Warning: {filename}: {exc}",
            file=sys.stderr,
        )


catalog = {

    "version": 1,

    "region": "europe",

    "generatedAt": (
        datetime.datetime.now(
            datetime.timezone.utc
        )
        .replace(microsecond=0)
        .isoformat()
    ),

    "count": len(regions),

    "packages": regions,
}


with output.open(
    "w",
    encoding="utf-8",
) as f:

    json.dump(
        catalog,
        f,
        ensure_ascii=False,
        indent=2,
    )


print(
    f"Catalogue : {len(regions)} package(s)"
)

PY


# ============================================================
# Résultat
# ============================================================

echo
echo "============================================================"
echo "                  NEXTLIMIT EUROPE"
echo "============================================================"
echo
echo "Détectés : ${TOTAL}"
echo "Construits : ${BUILT}"
echo "Déjà présents : ${SKIPPED}"
echo "Échecs : ${FAILED}"
echo
echo "Packages :"
echo "  ${OUTPUT_DIR}"
echo
echo "Catalogue :"
echo "  ${CATALOG_JSON}"
echo
echo "Rapport :"
echo "  ${SUMMARY_CSV}"
echo


if [[ "${FAILED}" -gt 0 ]]; then

    echo "Pays en erreur :"

    cat "${FAILED_FILE}"

    echo

fi

success "Traitement Europe terminé."

CONTAINER_SCRIPT


# ============================================================
# Vérification hôte
# ============================================================

CATALOG="${DATA_DIR}/catalog.json"

[[ -s "${CATALOG}" ]] \
    || die "Le catalogue final n'a pas été créé."


success "Datasets Europe générés."

echo

echo "Catalogue :"
echo "  ${CATALOG}"

echo

echo "Packages :"
du -sh "${DATA_DIR}/packages"

echo

echo "Nombre de PBF :"
find \
    "${DATA_DIR}/packages" \
    -maxdepth 1 \
    -name '*.osm.pbf' \
    -type f \
    | wc -l

echo

