diff --git a/.template.env b/.template.env
index 0eec464..86ff67f 100644
--- a/.template.env
+++ b/.template.env
@@ -6,9 +6,17 @@ SOURCE_STATS_CSV_PATH="https://www.data.gouv.fr/api/1/datasets/r/8ded94de-3b80-4
# Chemin vers le schéma de données
DATA_SCHEMA_PATH=https://www.data.gouv.fr/api/1/datasets/r/9a4144c0-ee44-4dec-bee5-bbef38191d9a
+# Colonnes masquées par défaut
+DISPLAYED_COLUMNS="uid, acheteur_id, acheteur_nom, montant, objet, titulaire_nom, titulaire_id, dateNotification, dureeMois, acheteur_departement_code, sourceDataset"
+
# Formulaire de contact
SENDER_SERVER_DOMAIN="mail.example.com" # serveur SMTP
LOGIN_PASSWORD="" # mot de passe du serveur
LOGIN_EMAIL="connect@example.fr" # adresse utilisée pour se connecter au serveur SMTP
FROM_EMAIL="from@example.com" # adresse d'envoi des emails (From)
TO_EMAIL="to@example.com" # adresse de destination des emails (To)
+
+# Matomo
+MATOMO_ID_SITE=
+MATOMO_BASE_URL=
+MATOMO_TOKEN=
diff --git a/README.md b/README.md
index 53a30f0..9f7e1de 100644
--- a/README.md
+++ b/README.md
@@ -38,6 +38,12 @@ Ne pas oublier de mettre à jour les fichier .env.
## Notes de version
+#### 2.2.0 (13 novembre 2025)
+
+- Moteur de recherche (acheteurs et titulaires) en page d'accueil ([#58](https://github.com/ColinMaudry/decp.info/issues/58))
+- Top acheteurs / titulaires par montant attribué/remporté (([#55](https://github.com/ColinMaudry/decp.info/issues/55)))
+- Moins de colonnes affichées par défaut dans Tableau ([#54](https://github.com/ColinMaudry/decp.info/issues/54))
+
##### 2.1.7 (11 novembre 2025)
- Remplacement du formulaire de contact par une adresse email
diff --git a/pyproject.toml b/pyproject.toml
index 5f29774..bfc01c5 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,7 +1,7 @@
[project]
name = "decp.info"
description = "Interface d'exploration et d'analyse des marchés publics français."
-version = "2.1.7"
+version = "2.2.0"
requires-python = ">= 3.10"
authors = [
{ name = "Colin Maudry", email = "colin+decp@maudry.com" }
@@ -16,7 +16,8 @@ dependencies = [
"xlsxwriter",
"plotly[express]",
"httpx",
- "pandas" # utilisé pour la création de certains graphiques
+ "pandas", # utilisé pour la création de certains graphiques
+ "unidecode"
]
[project.optional-dependencies]
diff --git a/src/assets/style.css b/src/assets/style.css
index 5626043..dd7cd60 100644
--- a/src/assets/style.css
+++ b/src/assets/style.css
@@ -53,9 +53,10 @@ div.logo > a {
/* Réduire la taille du texte de la colonne Objet */
-td[data-dash-column="objet"] {
+/*
+td[data-dash-column="objet"],td[data-dash-column="titulaire_nom"],td[data-dash-column="acheteur_nom"], {
font-size: 85%;
-}
+}*/
/* Couleur des en-têtes */
.dash-table-container
@@ -149,6 +150,35 @@ td[data-dash-column="objet"] {
font-size: 90%;
}
+/* Page de recherche */
+
+#search {
+ margin: 50px auto 0px auto;
+ width: 450px;
+ font-size: 18px;
+ height: 30px;
+ display: block;
+}
+
+.search_options {
+ margin: 16px auto;
+ width: 450px;
+}
+
+.search_options input {
+ margin-right: 12px;
+}
+
+.results_acheteur {
+ grid-column: 1;
+ grid-row: 1;
+}
+
+.results_titulaire {
+ grid-column: 2;
+ grid-row: 1;
+}
+
/* Menu de navigation */
.navbar {
@@ -178,7 +208,7 @@ summary > h3 {
padding-top: 28px;
}
-/* Vue acheteur/titulaire */
+/* Vue acheteur/titulaire/recherche */
.wrapper {
display: grid;
grid-gap: 10px;
@@ -186,9 +216,6 @@ summary > h3 {
justify-content: space-between;
}
-.wrapper > div {
-}
-
.org_title {
grid-column: 1 / 3;
grid-row: 1;
@@ -218,6 +245,11 @@ summary > h3 {
grid-row: 2;
}
+.org_top {
+ grid-column: 1/3;
+ grid-row: 3;
+}
+
/* Vue marché */
.marche_infos p {
diff --git a/src/callbacks.py b/src/callbacks.py
new file mode 100644
index 0000000..6bf197e
--- /dev/null
+++ b/src/callbacks.py
@@ -0,0 +1,65 @@
+import polars as pl
+from dash import dash_table, html
+
+from utils import add_links_in_dict, format_values, setup_table_columns
+
+
+def get_top_org_table(data, org_type: str):
+ dff = pl.DataFrame(data)
+ if dff.height == 0:
+ return html.Div()
+
+ dff = dff.select(
+ ["uid", f"{org_type}_id", f"{org_type}_nom", "distance", "montant"]
+ )
+ dff_nb = dff.group_by(f"{org_type}_id", f"{org_type}_nom", "distance").agg(
+ pl.len().alias("Attributions"), pl.sum("montant").alias("montant")
+ )
+ dff_nb = dff_nb.sort(by="montant", descending=True)
+ dff_nb = dff_nb.cast(pl.String)
+ dff_nb = dff_nb.fill_null("")
+ dff_nb = format_values(dff_nb)
+ columns, tooltip = setup_table_columns(
+ dff_nb, hideable=False, exclude=[f"{org_type}_id"]
+ )
+ data = dff_nb.to_dicts()
+ data = add_links_in_dict(data, f"{org_type}")
+
+ return dash_table.DataTable(
+ data=data,
+ markdown_options={"html": True},
+ page_action="native",
+ page_size=10,
+ columns=columns,
+ cell_selectable=False,
+ tooltip_header=tooltip,
+ style_cell_conditional=[
+ {
+ "if": {"column_id": "objet"},
+ "minWidth": "350px",
+ "textAlign": "left",
+ "overflow": "hidden",
+ "lineHeight": "14px",
+ "whiteSpace": "normal",
+ "fontSize": "85%",
+ },
+ {
+ "if": {"column_id": "acheteur_nom"},
+ "minWidth": "200px",
+ "textAlign": "left",
+ "overflow": "hidden",
+ "lineHeight": "16px",
+ # "fontSize": "85%",
+ "whiteSpace": "normal",
+ },
+ {
+ "if": {"column_id": "titulaire_nom"},
+ "minWidth": "200px",
+ "textAlign": "left",
+ "overflow": "hidden",
+ "lineHeight": "16px",
+ "whiteSpace": "normal",
+ # "fontSize": "85%",
+ },
+ ],
+ )
diff --git a/src/pages/acheteur.py b/src/pages/acheteur.py
index 80a258f..87ddd05 100644
--- a/src/pages/acheteur.py
+++ b/src/pages/acheteur.py
@@ -3,12 +3,13 @@ import datetime
import polars as pl
from dash import Input, Output, State, callback, dash_table, dcc, html, register_page
+from src.callbacks import get_top_org_table
from src.figures import point_on_map
from src.utils import (
add_links_in_dict,
df,
- format_montant,
format_number,
+ format_values,
get_annuaire_data,
get_departement_region,
meta_content,
@@ -89,6 +90,13 @@ layout = [
],
),
html.Div(className="org_map", id="acheteur_map"),
+ html.Div(
+ className="org_top",
+ children=[
+ html.H3("Top titulaires"),
+ html.Div(className="marches_table", id="top10_titulaires"),
+ ],
+ ),
],
),
# récupérer les données de l'acheteur sur l'api annuaire
@@ -146,22 +154,22 @@ def update_acheteur_infos(url):
Input(component_id="acheteur_data", component_property="data"),
)
def update_acheteur_stats(data):
- df = pl.DataFrame(data)
- if df.height == 0:
- df = pl.DataFrame(schema=df.collect_schema())
- df_marches = df.unique("id")
+ dff = pl.DataFrame(data)
+ if dff.height == 0:
+ dff = pl.DataFrame(schema=df.collect_schema())
+ df_marches = dff.unique("id")
nb_marches = format_number(df_marches.height)
# somme_marches = format_number(int(df_marches.select(pl.sum("montant")).item()))
marches_attribues = [html.Strong(nb_marches), " marchés et accord-cadres attribués"]
# + ", pour un total de ", html.Strong(somme_marches + " €")]
del df_marches
- nb_titulaires = df.unique("titulaire_id").height
+ nb_titulaires = dff.unique("titulaire_id").height
nb_titulaires = [
html.Strong(format_number(nb_titulaires)),
" titulaires (SIRET) différents",
]
- del df
+ del dff
return marches_attribues, nb_titulaires
@@ -183,6 +191,7 @@ def get_acheteur_marches_data(url, acheteur_year: str) -> list[dict]:
"titulaire_id",
"titulaire_typeIdentifiant",
"titulaire_nom",
+ "distance",
"montant",
"codeCPV",
"dureeMois",
@@ -203,9 +212,11 @@ def get_acheteur_marches_data(url, acheteur_year: str) -> list[dict]:
)
def get_last_marches_table(data) -> html.Div:
dff = pl.DataFrame(data)
+ if dff.height == 0:
+ return html.Div(html.P("Aucun marché trouvé."))
dff = dff.cast(pl.String)
dff = dff.fill_null("")
- dff = format_montant(dff)
+ dff = format_values(dff)
columns, tooltip = setup_table_columns(
dff,
hideable=False,
@@ -235,7 +246,7 @@ def get_last_marches_table(data) -> html.Div:
"minWidth": "300px",
"textAlign": "left",
"overflow": "hidden",
- "lineHeight": "14px",
+ "lineHeight": "18px",
"whiteSpace": "normal",
},
{
@@ -252,6 +263,14 @@ def get_last_marches_table(data) -> html.Div:
return table
+@callback(
+ Output(component_id="top10_titulaires", component_property="children"),
+ Input(component_id="acheteur_data", component_property="data"),
+)
+def get_top_titulaires(data):
+ return get_top_org_table(data, "titulaire")
+
+
@callback(
Output("download-acheteur-data", "data"),
Input("btn-download-acheteur-data", "n_clicks"),
diff --git a/src/pages/marche.py b/src/pages/marche.py
index bd90a50..807bb7a 100644
--- a/src/pages/marche.py
+++ b/src/pages/marche.py
@@ -4,7 +4,7 @@ import polars as pl
from dash import Input, Output, callback, dcc, html, register_page
from polars import selectors as cs
-from src.utils import data_schema, df, format_montant, meta_content
+from src.utils import data_schema, df, format_values, meta_content
register_page(
__name__,
@@ -75,7 +75,7 @@ def get_marche_data(url) -> tuple[dict, list]:
# Données du marché
dff_marche = lff.unique("uid").collect(engine="streaming")
- dff_marche = format_montant(dff_marche)
+ dff_marche = format_values(dff_marche)
return dff_marche.to_dicts()[0], dff_titulaires.to_dicts()
diff --git a/src/pages/recherche.py b/src/pages/recherche.py
new file mode 100644
index 0000000..b8b4796
--- /dev/null
+++ b/src/pages/recherche.py
@@ -0,0 +1,105 @@
+from dash import Input, Output, callback, dash_table, dcc, html, register_page
+
+from src.utils import (
+ df_acheteurs,
+ df_titulaires,
+ meta_content,
+ search_org,
+ setup_table_columns,
+)
+
+name = "Recherche"
+
+register_page(
+ __name__,
+ path="/",
+ title=meta_content["title"],
+ name=name,
+ description=meta_content["description"],
+ image_url=meta_content["image_url"],
+ order=0,
+)
+
+layout = html.Div(
+ className="container",
+ children=[
+ dcc.Input(
+ id="search",
+ type="text",
+ placeholder="Nom d'acheteur, d'entreprise, SIREN...",
+ autoFocus=True,
+ ),
+ # html.Div(
+ # className="search_options",
+ # children=[dcc.RadioItems(options=["Acheteur(s)"])],
+ # ),
+ html.Div(id="search_results", className="wrapper"),
+ ],
+)
+
+
+@callback(
+ Output("search_results", "children"),
+ Input("search", "value"),
+ prevent_initial_call=True,
+)
+def update_search_results(query):
+ if len(query) >= 1:
+ content = []
+
+ for org_type in ["acheteur", "titulaire"]:
+ if org_type == "acheteur":
+ dff = df_acheteurs
+ elif org_type == "titulaire":
+ dff = df_titulaires
+ else:
+ raise ValueError(f"{org_type} is not supported")
+
+ # Search acheteurs and titulaires using the same function
+ results = search_org(dff, query, org_type=org_type)
+ count = results.height
+
+ # Format output
+ columns, tooltip = setup_table_columns(results, hideable=False)
+
+ org_content = [
+ html.Div(
+ className=f"results_{org_type}",
+ children=[
+ html.H3(f"{org_type.title()}s : {count}"),
+ dash_table.DataTable(
+ columns=columns,
+ data=results.to_dicts(),
+ page_size=10,
+ # style_table={"overflowX": "auto"},
+ markdown_options={"html": True},
+ cell_selectable=False,
+ style_cell_conditional=[
+ {
+ "if": {"column_id": "acheteur_nom"},
+ "maxWidth": "250px",
+ "textAlign": "left",
+ "overflow": "hidden",
+ "lineHeight": "18px",
+ "whiteSpace": "normal",
+ },
+ {
+ "if": {"column_id": "titulaire_nom"},
+ "maxWidth": "250px",
+ "textAlign": "left",
+ "overflow": "hidden",
+ "lineHeight": "18px",
+ "whiteSpace": "normal",
+ },
+ ],
+ ),
+ ],
+ )
+ if count > 0
+ else html.P(f"Aucun {org_type} trouvé."),
+ ]
+ content.extend(org_content)
+
+ return content
+ else:
+ return html.P("")
diff --git a/src/pages/tableau.py b/src/pages/tableau.py
index 23de881..972a240 100644
--- a/src/pages/tableau.py
+++ b/src/pages/tableau.py
@@ -9,8 +9,9 @@ from src.utils import (
add_resource_link,
df,
filter_table_data,
- format_montant,
format_number,
+ format_values,
+ get_default_hidden_columns,
meta_content,
setup_table_columns,
sort_table_data,
@@ -24,7 +25,7 @@ schema = df.collect_schema()
name = "Tableau"
register_page(
__name__,
- path="/",
+ path="/tableau",
title=meta_content["title"],
name=name,
description=meta_content["description"],
@@ -52,7 +53,7 @@ datatable = html.Div(
"minWidth": "350px",
"textAlign": "left",
"overflow": "hidden",
- "lineHeight": "14px",
+ "lineHeight": "18px",
"whiteSpace": "normal",
},
{
@@ -60,7 +61,7 @@ datatable = html.Div(
"minWidth": "250px",
"textAlign": "left",
"overflow": "hidden",
- "lineHeight": "14px",
+ "lineHeight": "18px",
"whiteSpace": "normal",
},
{
@@ -68,7 +69,7 @@ datatable = html.Div(
"minWidth": "250px",
"textAlign": "left",
"overflow": "hidden",
- "lineHeight": "14px",
+ "lineHeight": "18px",
"whiteSpace": "normal",
},
],
@@ -76,6 +77,7 @@ datatable = html.Div(
markdown_options={"html": True},
tooltip_duration=8000,
tooltip_delay=350,
+ hidden_columns=get_default_hidden_columns(schema),
),
)
@@ -206,7 +208,7 @@ def update_table(page_current, page_size, filter_query, sort_by, data_timestamp)
dff = add_resource_link(dff)
# Formatage des montants
- dff = format_montant(dff)
+ dff = format_values(dff)
columns, tooltip = setup_table_columns(dff)
diff --git a/src/pages/titulaire.py b/src/pages/titulaire.py
index 07f07eb..3ebeaef 100644
--- a/src/pages/titulaire.py
+++ b/src/pages/titulaire.py
@@ -3,12 +3,13 @@ import datetime
import polars as pl
from dash import Input, Output, State, callback, dash_table, dcc, html, register_page
+from src.callbacks import get_top_org_table
from src.figures import point_on_map
from src.utils import (
add_links_in_dict,
df,
- format_montant,
format_number,
+ format_values,
get_annuaire_data,
get_departement_region,
meta_content,
@@ -91,6 +92,13 @@ layout = [
],
),
html.Div(className="org_map", id="titulaire_map"),
+ html.Div(
+ className="org_top",
+ children=[
+ html.H3("Top acheteurs"),
+ html.Div(className="marches_table", id="top10_acheteurs"),
+ ],
+ ),
],
),
# récupérer les données de l'acheteur sur l'api annuaire
@@ -161,7 +169,7 @@ def update_titulaire_stats(data):
nb_acheteurs = dff.unique("acheteur_id").height
nb_acheteurs = [
html.Strong(format_number(nb_acheteurs)),
- " titulaires (SIRET) différents",
+ " acheteurs (SIRET) différents",
]
del dff
@@ -187,6 +195,7 @@ def get_titulaire_marches_data(url, titulaire_year: str) -> list[dict]:
"dateNotification",
"acheteur_id",
"acheteur_nom",
+ "distance",
"montant",
"codeCPV",
"dureeMois",
@@ -219,7 +228,7 @@ def get_last_marches_table(data) -> html.Div:
dff = pl.DataFrame(data)
dff = dff.cast(pl.String)
dff = dff.fill_null("")
- dff = format_montant(dff)
+ dff = format_values(dff)
columns, tooltip = setup_table_columns(
dff, hideable=False, exclude=["acheteur_id", "id"]
)
@@ -249,7 +258,7 @@ def get_last_marches_table(data) -> html.Div:
"minWidth": "300px",
"textAlign": "left",
"overflow": "hidden",
- "lineHeight": "14px",
+ "lineHeight": "18px",
"whiteSpace": "normal",
},
{
@@ -266,6 +275,14 @@ def get_last_marches_table(data) -> html.Div:
return table
+@callback(
+ Output(component_id="top10_acheteurs", component_property="children"),
+ Input(component_id="titulaire_data", component_property="data"),
+)
+def get_top_acheteurs(data):
+ return get_top_org_table(data, "acheteur")
+
+
@callback(
Output("download-titulaire-data", "data"),
Input("btn-download-titulaire-data", "n_clicks"),
diff --git a/src/utils.py b/src/utils.py
index 85d9e43..eaaf9a5 100644
--- a/src/utils.py
+++ b/src/utils.py
@@ -1,22 +1,19 @@
import json
import logging
import os
-from time import sleep
+import uuid
+from time import localtime, sleep
import polars as pl
import polars.selectors as cs
-from httpx import get
+from httpx import get, post
+from polars import Schema
from polars.exceptions import ComputeError
-
-operators = [
- ["s<", "<"],
- ["s>", ">"],
- ["i<", "<"],
- ["i>", ">"],
- ["icontains", "contains"],
-]
+from unidecode import unidecode
logger = logging.getLogger("decp.info")
+logging.getLogger("httpx").setLevel("WARNING")
+
logging.basicConfig(
format="%(asctime)s %(levelname)-8s %(message)s",
level=logging.INFO,
@@ -25,6 +22,13 @@ logging.basicConfig(
def split_filter_part(filter_part):
+ operators = [
+ ["s<", "<"],
+ ["s>", ">"],
+ ["i<", "<"],
+ ["i>", ">"],
+ ["icontains", "contains"],
+ ]
print("filter part", filter_part)
for operator_group in operators:
if operator_group[0] in filter_part:
@@ -49,42 +53,60 @@ def add_resource_link(dff: pl.DataFrame) -> pl.DataFrame:
return dff
-def add_links(dff: pl.DataFrame):
- dff = dff.with_columns(
- pl.when(pl.col("titulaire_typeIdentifiant") == "SIRET")
- .then(
- ''
- + pl.col("titulaire_id")
- + ""
- )
- .otherwise(pl.col("titulaire_id"))
- .alias("titulaire_id")
- )
-
- for column, path in [("acheteur_id", "acheteurs"), ("uid", "marches")]:
- dff = dff.with_columns(
- (
- f''
- + pl.col(column)
- + ""
- ).alias(column)
- )
+def add_links(dff: pl.DataFrame, target: str = "_blank"):
+ for col in ["uid", "acheteur_nom", "titulaire_nom", "acheteur_id", "titulaire_id"]:
+ if col in dff.columns:
+ if col.startswith("titulaire_"):
+ dff = dff.with_columns(
+ pl.when(
+ pl.Expr.or_(
+ pl.col("titulaire_typeIdentifiant").is_null(),
+ pl.col("titulaire_typeIdentifiant") == "SIRET",
+ )
+ )
+ .then(
+ ''
+ + pl.col(col)
+ + ""
+ )
+ .otherwise(pl.col(col))
+ .alias(col)
+ )
+ if col.startswith("acheteur_"):
+ dff = dff.with_columns(
+ (
+ ''
+ + pl.col(col)
+ + ""
+ ).alias(col)
+ )
+ if col == "uid":
+ dff = dff.with_columns(
+ (
+ ''
+ + pl.col("uid")
+ + ""
+ ).alias("uid")
+ )
return dff
-def add_links_in_dict(data: list, org_type: str) -> list:
+def add_links_in_dict(data: list[dict], org_type: str) -> list:
new_data = []
for marche in data:
org_id = marche[org_type + "_id"]
marche[org_type + "_nom"] = (
f'{marche[org_type + "_nom"]}'
)
- marche["id"] = f'{marche["id"]}'
- marche["uid"] = f'{marche["uid"]}'
+ if marche.get("uid"):
+ marche["id"] = f'{marche["id"]}'
+ marche["uid"] = f'{marche["uid"]}'
new_data.append(marche)
return new_data
@@ -123,8 +145,8 @@ def format_number(number) -> str:
return number
-def format_montant(dff: pl.DataFrame) -> pl.DataFrame:
- def format_function(expr, scale=None):
+def format_values(dff: pl.DataFrame) -> pl.DataFrame:
+ def format_montant(expr, scale=None):
# https://stackoverflow.com/a/78636786
expr = expr.cast(pl.String)
expr = expr.str.splitn(".", 2)
@@ -142,7 +164,7 @@ def format_montant(dff: pl.DataFrame) -> pl.DataFrame:
frac: pl.Expr = (
pl.when(frac.is_not_null() & ~frac.is_in(["0"]))
- .then("," + frac)
+ .then("," + frac.str.head(2))
.otherwise(pl.lit(""))
)
@@ -154,7 +176,17 @@ def format_montant(dff: pl.DataFrame) -> pl.DataFrame:
return montant
- dff = dff.with_columns(pl.col("montant").pipe(format_function).alias("montant"))
+ def format_distance(expr):
+ expr = expr.cast(pl.String)
+ return pl.concat_str(expr, pl.lit(" km"))
+
+ if "montant" in dff.columns:
+ dff = dff.with_columns(pl.col("montant").pipe(format_montant).alias("montant"))
+ if "distance" in dff.columns:
+ dff = dff.with_columns(
+ pl.col("distance").pipe(format_distance).alias("distance")
+ )
+
return dff
@@ -196,6 +228,18 @@ def get_decp_data() -> pl.DataFrame:
return lff.collect()
+def get_org_data(dff: pl.DataFrame, org_type: str) -> pl.DataFrame:
+ lff = dff.lazy()
+ lff = lff.select(
+ "uid",
+ cs.starts_with(org_type).exclude(
+ f"{org_type}_latitude", f"{org_type}_longitude"
+ ),
+ )
+ lff = lff.group_by(cs.starts_with(org_type)).len("Marchés")
+ return lff.collect()
+
+
def get_departements() -> dict:
with open("data/departements.json", "rb") as f:
data = json.load(f)
@@ -312,6 +356,20 @@ def setup_table_columns(dff, hideable: bool = True, exclude: list = None) -> tup
return columns, tooltip
+def get_default_hidden_columns(schema: Schema):
+ displayed_columns = os.getenv("DISPLAYED_COLUMNS")
+ hidden_columns = []
+ if displayed_columns:
+ displayed_columns = displayed_columns.replace(" ", "").split(",")
+ for col in schema.names():
+ if col in displayed_columns:
+ continue
+ else:
+ hidden_columns.append(col)
+ return hidden_columns
+ raise ValueError("DISPLAYED_COLUMNS n'est pas configuré")
+
+
def get_data_schema() -> dict:
# Récupération du schéma des données tabulaires
path = os.getenv("DATA_SCHEMA_PATH")
@@ -338,7 +396,114 @@ def get_data_schema() -> dict:
return new_schema
+def track_search(query):
+ if (
+ len(query) >= 4
+ and os.getenv("DEVELOPMENT").lower != "true"
+ and os.getenv("MATOMO_DOMAIN")
+ ):
+ if os.getenv("DEVELOPMENT").lower() == "true":
+ url = "https://test.decp.info"
+ else:
+ url = "https://decp.info"
+ params = {
+ "idsite": os.getenv("MATOMO_ID_SITE"),
+ "url": url,
+ "rec": "1",
+ "action_name": "front_page_search",
+ "rand": uuid.uuid4().hex,
+ "apiv": "1",
+ "h": localtime().tm_hour,
+ "m": localtime().tm_min,
+ "s": localtime().tm_sec,
+ "search": query,
+ "token_auth": os.getenv("MATOMO_TOKEN"),
+ }
+ post(
+ url=f"https://{os.getenv('MATOMO_DOMAIN')}/matomo.php",
+ params=params,
+ ).raise_for_status()
+
+
+def search_org(dff: pl.DataFrame, query: str, org_type: str) -> pl.DataFrame:
+ """
+ Search in either 'acheteur' or 'titulaire' DataFrame.
+
+ :param dff: Polars DataFrame with acheteur or titulaire columns
+ :param query: User search string
+ :param org_type: 'acheteur' or 'titulaire'
+ :return: Filtered DataFrame with 'matches' column
+ """
+ if not query.strip():
+ return dff.select(pl.lit(False).alias("matches"))
+
+ # Enregistrement des recherche dans Matomo
+ track_search(query)
+
+ # Normalize query
+ normalized_query = unidecode(query.strip()).upper()
+ tokens = [" " + t.strip() for t in normalized_query.split() if t.strip()]
+
+ # Define columns based on entity type
+ cols = [
+ f"{org_type}_id",
+ f"{org_type}_nom",
+ f"{org_type}_departement_nom",
+ f"{org_type}_departement_code",
+ f"{org_type}_commune_nom",
+ ]
+
+ # Concatenate all fields into one string per row
+ org_str = pl.concat_str(pl.lit(" "), pl.col(cols), separator=" ")
+
+ # For each token, create a boolean column: True if token is found
+ token_matches = []
+ for token in tokens:
+ token_match = org_str.str.contains(token).alias(f"token_{token}")
+ token_matches.append(token_match)
+
+ # Count how many tokens match per row
+ match_score = pl.sum_horizontal(token_matches).alias("match_score")
+
+ # For each token, create a boolean column: True if token is found
+ token_matches = []
+ for token in tokens:
+ token_match = org_str.str.contains(token).alias(f"token_{token}")
+ token_matches.append(token_match)
+
+ # Sélection des colonnes
+ if org_type == "acheteur":
+ dff = dff.select(cols + ["Marchés"])
+ if org_type == "titulaire":
+ dff = dff.select(cols + ["Marchés", "titulaire_typeIdentifiant"])
+
+ # Apply and filter
+ dff = (
+ dff.with_columns(token_matches + [match_score])
+ .filter(pl.col("match_score") == len(tokens))
+ .sort("Marchés", descending=True)
+ .drop([f"token_{token}" for token in tokens])
+ )
+
+ # Format result
+ dff = add_links(dff, target="")
+ dff = dff.with_columns(
+ pl.concat_str(
+ pl.col(f"{org_type}_departement_nom"),
+ pl.lit(" ("),
+ pl.col(f"{org_type}_departement_code"),
+ pl.lit(")"),
+ ).alias("Département")
+ )
+
+ dff = dff.select(f"{org_type}_id", f"{org_type}_nom", "Département", "Marchés")
+
+ return dff
+
+
df: pl.DataFrame = get_decp_data()
+df_acheteurs = get_org_data(df, "acheteur")
+df_titulaires = get_org_data(df, "titulaire")
departements = get_departements()
domain_name = (
"test.decp.info" if os.getenv("DEVELOPMENT").lower() == "true" else "decp.info"