Possibilité de charger les données depuis une URL, memoisation de last_modified
This commit is contained in:
+1
-1
@@ -42,6 +42,6 @@ env = [
|
|||||||
"DATA_FILE_PARQUET_PATH=tests/test.parquet",
|
"DATA_FILE_PARQUET_PATH=tests/test.parquet",
|
||||||
"DEVELOPMENT=true",
|
"DEVELOPMENT=true",
|
||||||
"REBUILD_DUCKDB=true",
|
"REBUILD_DUCKDB=true",
|
||||||
"DATA_SCHEMA_PATH=/home/colin/git/decp-processing/dist/schema.json",
|
"DATA_SCHEMA_LOCAL=/home/colin/git/decp-processing/dist/schema.json",
|
||||||
]
|
]
|
||||||
addopts = "-p no:warnings"
|
addopts = "-p no:warnings"
|
||||||
|
|||||||
+14
-8
@@ -6,7 +6,7 @@ import pandas # noqa: F401 # eager import: avoid plotly's lazy-import race acr
|
|||||||
import tomllib
|
import tomllib
|
||||||
from dash import Dash, Input, Output, State, dcc, html, page_container, page_registry
|
from dash import Dash, Input, Output, State, dcc, html, page_container, page_registry
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from flask import Response
|
from flask import Flask, Response
|
||||||
|
|
||||||
from src.utils import DEVELOPMENT
|
from src.utils import DEVELOPMENT
|
||||||
from src.utils.cache import cache
|
from src.utils.cache import cache
|
||||||
@@ -27,12 +27,10 @@ META_TAGS = [
|
|||||||
if DEVELOPMENT:
|
if DEVELOPMENT:
|
||||||
META_TAGS.append({"name": "robots", "content": "noindex"})
|
META_TAGS.append({"name": "robots", "content": "noindex"})
|
||||||
|
|
||||||
app: Dash = Dash(
|
# Le cache doit être initialisé AVANT la construction de Dash : `use_pages=True`
|
||||||
title="decp.info",
|
# importe les modules de pages pendant l'instanciation, et certains appellent des
|
||||||
use_pages=True,
|
# fonctions memoizées (@cache.memoize) dès l'import (ex. tableau.py).
|
||||||
compress=True,
|
server = Flask(__name__)
|
||||||
meta_tags=META_TAGS,
|
|
||||||
)
|
|
||||||
|
|
||||||
cache_dir = os.getenv("CACHE_DIR", "/tmp/decp-cache")
|
cache_dir = os.getenv("CACHE_DIR", "/tmp/decp-cache")
|
||||||
|
|
||||||
@@ -40,7 +38,7 @@ if os.path.exists(cache_dir):
|
|||||||
rmtree(cache_dir)
|
rmtree(cache_dir)
|
||||||
|
|
||||||
cache.init_app(
|
cache.init_app(
|
||||||
app.server,
|
server,
|
||||||
config={
|
config={
|
||||||
"CACHE_TYPE": "FileSystemCache",
|
"CACHE_TYPE": "FileSystemCache",
|
||||||
"CACHE_DIR": cache_dir,
|
"CACHE_DIR": cache_dir,
|
||||||
@@ -51,6 +49,14 @@ cache.init_app(
|
|||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
app: Dash = Dash(
|
||||||
|
server=server,
|
||||||
|
title="decp.info",
|
||||||
|
use_pages=True,
|
||||||
|
compress=True,
|
||||||
|
meta_tags=META_TAGS,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
# robots.txt
|
# robots.txt
|
||||||
@app.server.route("/robots.txt")
|
@app.server.route("/robots.txt")
|
||||||
|
|||||||
@@ -8,27 +8,31 @@ import polars as pl
|
|||||||
import polars.selectors as cs
|
import polars.selectors as cs
|
||||||
from polars.exceptions import ComputeError
|
from polars.exceptions import ComputeError
|
||||||
|
|
||||||
from src.utils import logger
|
from src.utils import get_last_modified, logger
|
||||||
|
|
||||||
|
|
||||||
def should_rebuild(db_path: Path, parquet_path: Path) -> bool:
|
def should_rebuild(db_path: Path, parquet_path: str) -> bool:
|
||||||
db_path = Path(db_path)
|
db_path = Path(db_path)
|
||||||
parquet_path = Path(parquet_path)
|
|
||||||
if not db_path.exists():
|
if not db_path.exists():
|
||||||
return True
|
return True
|
||||||
dev = os.getenv("DEVELOPMENT", "False").lower() == "true"
|
dev = os.getenv("DEVELOPMENT", "False").lower() == "true"
|
||||||
force = os.getenv("REBUILD_DUCKDB", "False").lower() == "true"
|
force = os.getenv("REBUILD_DUCKDB", "False").lower() == "true"
|
||||||
if dev and not force:
|
if dev and not force:
|
||||||
return False
|
return False
|
||||||
return parquet_path.stat().st_mtime > db_path.stat().st_mtime
|
last_modified: float = get_last_modified(parquet_path)
|
||||||
|
return last_modified > db_path.stat().st_mtime
|
||||||
|
|
||||||
|
|
||||||
def _load_source_frame(parquet_path: Path) -> pl.DataFrame:
|
def _load_source_frame() -> pl.DataFrame:
|
||||||
"""Read the source parquet and apply the row-level transforms.
|
"""Read the source parquet and apply the row-level transforms.
|
||||||
|
|
||||||
Kept here (not in utils.py) so src.db has no dependency on utils.
|
Kept here (not in utils.py) so src.db has no dependency on utils.
|
||||||
Mirrors the behavior previously in utils.get_decp_data().
|
Mirrors the behavior previously in utils.get_decp_data().
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
parquet_path: str = os.getenv("DATA_FILE_PARQUET_PATH", "")
|
||||||
|
if not (parquet_path.startswith("http")):
|
||||||
|
assert os.path.exists(parquet_path)
|
||||||
try:
|
try:
|
||||||
lff: pl.LazyFrame = pl.scan_parquet(str(parquet_path))
|
lff: pl.LazyFrame = pl.scan_parquet(str(parquet_path))
|
||||||
except ComputeError:
|
except ComputeError:
|
||||||
@@ -58,20 +62,21 @@ def _load_source_frame(parquet_path: Path) -> pl.DataFrame:
|
|||||||
return lff.collect()
|
return lff.collect()
|
||||||
|
|
||||||
|
|
||||||
def build_database(db_path: Path, parquet_path: Path) -> None:
|
def build_database(db_path: Path) -> None:
|
||||||
"""Build the DuckDB database atomically under an exclusive lock.
|
"""Build the DuckDB database atomically under an exclusive lock.
|
||||||
|
|
||||||
Caller MUST hold the fcntl.flock on the .lock file.
|
Caller MUST hold the fcntl.flock on the .lock file.
|
||||||
"""
|
"""
|
||||||
db_path = Path(db_path)
|
db_path = Path(db_path)
|
||||||
parquet_path = Path(parquet_path)
|
|
||||||
tmp_path = db_path.with_suffix(".duckdb.tmp")
|
tmp_path = db_path.with_suffix(".duckdb.tmp")
|
||||||
staging_parquet = db_path.with_suffix(".staging.parquet")
|
staging_parquet = db_path.with_suffix(".staging.parquet")
|
||||||
if tmp_path.exists():
|
if tmp_path.exists():
|
||||||
tmp_path.unlink()
|
tmp_path.unlink()
|
||||||
|
|
||||||
logger.info(f"Construction de la base DuckDB à partir de {parquet_path}...")
|
logger.info(
|
||||||
frame = _load_source_frame(parquet_path)
|
f"Construction de la base DuckDB à partir de {os.getenv('DATA_FILE_PARQUET_PATH', '')}..."
|
||||||
|
)
|
||||||
|
frame = _load_source_frame()
|
||||||
|
|
||||||
# Write transformed frame as parquet so DuckDB can read it natively
|
# Write transformed frame as parquet so DuckDB can read it natively
|
||||||
# (avoids pyarrow dependency for the Polars→DuckDB handoff)
|
# (avoids pyarrow dependency for the Polars→DuckDB handoff)
|
||||||
@@ -111,13 +116,13 @@ def build_database(db_path: Path, parquet_path: Path) -> None:
|
|||||||
|
|
||||||
def _ensure_database() -> Path:
|
def _ensure_database() -> Path:
|
||||||
db_path = Path(os.getenv("DUCKDB_PATH", "./decp.duckdb"))
|
db_path = Path(os.getenv("DUCKDB_PATH", "./decp.duckdb"))
|
||||||
parquet_path = Path(os.getenv("DATA_FILE_PARQUET_PATH"))
|
parquet_path = os.getenv("DATA_FILE_PARQUET_PATH", "")
|
||||||
lock_path = db_path.with_suffix(".duckdb.lock")
|
lock_path = db_path.with_suffix(".duckdb.lock")
|
||||||
|
|
||||||
with open(lock_path, "w") as lock_fd:
|
with open(lock_path, "w") as lock_fd:
|
||||||
fcntl.flock(lock_fd, fcntl.LOCK_EX)
|
fcntl.flock(lock_fd, fcntl.LOCK_EX)
|
||||||
if should_rebuild(db_path, parquet_path):
|
if should_rebuild(db_path, parquet_path):
|
||||||
build_database(db_path, parquet_path)
|
build_database(db_path)
|
||||||
else:
|
else:
|
||||||
logger.debug("Base de données déjà disponible et à jour.")
|
logger.debug("Base de données déjà disponible et à jour.")
|
||||||
return db_path
|
return db_path
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ from dash import (
|
|||||||
|
|
||||||
from src.db import query_marches, schema
|
from src.db import query_marches, schema
|
||||||
from src.figures import DataTable, make_column_picker
|
from src.figures import DataTable, make_column_picker
|
||||||
from src.utils import logger
|
from src.utils import get_last_modified, logger
|
||||||
from src.utils.seo import META_CONTENT
|
from src.utils.seo import META_CONTENT
|
||||||
from src.utils.table import (
|
from src.utils.table import (
|
||||||
COLUMNS,
|
COLUMNS,
|
||||||
@@ -33,7 +33,7 @@ from src.utils.table import (
|
|||||||
)
|
)
|
||||||
from src.utils.tracking import track_search
|
from src.utils.tracking import track_search
|
||||||
|
|
||||||
update_date_timestamp = os.path.getmtime(os.getenv("DATA_FILE_PARQUET_PATH"))
|
update_date_timestamp = get_last_modified(os.getenv("DATA_FILE_PARQUET_PATH", ""))
|
||||||
update_date = datetime.fromtimestamp(update_date_timestamp).strftime("%d/%m/%Y")
|
update_date = datetime.fromtimestamp(update_date_timestamp).strftime("%d/%m/%Y")
|
||||||
update_date_iso = datetime.fromtimestamp(update_date_timestamp).isoformat()
|
update_date_iso = datetime.fromtimestamp(update_date_timestamp).isoformat()
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,29 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from src.utils.cache import cache
|
||||||
|
|
||||||
|
|
||||||
|
@cache.memoize()
|
||||||
|
def get_last_modified(parquet_path: str) -> float:
|
||||||
|
logger.info("Récupération de la date de modification des données...")
|
||||||
|
logging.getLogger("httpx").setLevel("WARNING")
|
||||||
|
if parquet_path.startswith("http"):
|
||||||
|
last_modified = httpx.head(
|
||||||
|
url=parquet_path,
|
||||||
|
follow_redirects=True,
|
||||||
|
).headers["last-modified"]
|
||||||
|
last_modified = datetime.strptime(last_modified, "%a, %d %b %Y %X %Z").strftime(
|
||||||
|
"%s"
|
||||||
|
)
|
||||||
|
return float(last_modified)
|
||||||
|
parquet_local_path = Path(parquet_path)
|
||||||
|
return parquet_local_path.stat().st_mtime
|
||||||
|
|
||||||
|
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
format="%(asctime)s %(levelname)-8s %(message)s",
|
format="%(asctime)s %(levelname)-8s %(message)s",
|
||||||
|
|||||||
Reference in New Issue
Block a user