diff --git a/.template.env b/.template.env index 0eec464..86ff67f 100644 --- a/.template.env +++ b/.template.env @@ -6,9 +6,17 @@ SOURCE_STATS_CSV_PATH="https://www.data.gouv.fr/api/1/datasets/r/8ded94de-3b80-4 # Chemin vers le schéma de données DATA_SCHEMA_PATH=https://www.data.gouv.fr/api/1/datasets/r/9a4144c0-ee44-4dec-bee5-bbef38191d9a +# Colonnes masquées par défaut +DISPLAYED_COLUMNS="uid, acheteur_id, acheteur_nom, montant, objet, titulaire_nom, titulaire_id, dateNotification, dureeMois, acheteur_departement_code, sourceDataset" + # Formulaire de contact SENDER_SERVER_DOMAIN="mail.example.com" # serveur SMTP LOGIN_PASSWORD="" # mot de passe du serveur LOGIN_EMAIL="connect@example.fr" # adresse utilisée pour se connecter au serveur SMTP FROM_EMAIL="from@example.com" # adresse d'envoi des emails (From) TO_EMAIL="to@example.com" # adresse de destination des emails (To) + +# Matomo +MATOMO_ID_SITE= +MATOMO_BASE_URL= +MATOMO_TOKEN= diff --git a/README.md b/README.md index 53a30f0..9f7e1de 100644 --- a/README.md +++ b/README.md @@ -38,6 +38,12 @@ Ne pas oublier de mettre à jour les fichier .env. ## Notes de version +#### 2.2.0 (13 novembre 2025) + +- Moteur de recherche (acheteurs et titulaires) en page d'accueil ([#58](https://github.com/ColinMaudry/decp.info/issues/58)) +- Top acheteurs / titulaires par montant attribué/remporté (([#55](https://github.com/ColinMaudry/decp.info/issues/55))) +- Moins de colonnes affichées par défaut dans Tableau ([#54](https://github.com/ColinMaudry/decp.info/issues/54)) + ##### 2.1.7 (11 novembre 2025) - Remplacement du formulaire de contact par une adresse email diff --git a/pyproject.toml b/pyproject.toml index 5f29774..bfc01c5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "decp.info" description = "Interface d'exploration et d'analyse des marchés publics français." -version = "2.1.7" +version = "2.2.0" requires-python = ">= 3.10" authors = [ { name = "Colin Maudry", email = "colin+decp@maudry.com" } @@ -16,7 +16,8 @@ dependencies = [ "xlsxwriter", "plotly[express]", "httpx", - "pandas" # utilisé pour la création de certains graphiques + "pandas", # utilisé pour la création de certains graphiques + "unidecode" ] [project.optional-dependencies] diff --git a/src/assets/style.css b/src/assets/style.css index 5626043..dd7cd60 100644 --- a/src/assets/style.css +++ b/src/assets/style.css @@ -53,9 +53,10 @@ div.logo > a { /* Réduire la taille du texte de la colonne Objet */ -td[data-dash-column="objet"] { +/* +td[data-dash-column="objet"],td[data-dash-column="titulaire_nom"],td[data-dash-column="acheteur_nom"], { font-size: 85%; -} +}*/ /* Couleur des en-têtes */ .dash-table-container @@ -149,6 +150,35 @@ td[data-dash-column="objet"] { font-size: 90%; } +/* Page de recherche */ + +#search { + margin: 50px auto 0px auto; + width: 450px; + font-size: 18px; + height: 30px; + display: block; +} + +.search_options { + margin: 16px auto; + width: 450px; +} + +.search_options input { + margin-right: 12px; +} + +.results_acheteur { + grid-column: 1; + grid-row: 1; +} + +.results_titulaire { + grid-column: 2; + grid-row: 1; +} + /* Menu de navigation */ .navbar { @@ -178,7 +208,7 @@ summary > h3 { padding-top: 28px; } -/* Vue acheteur/titulaire */ +/* Vue acheteur/titulaire/recherche */ .wrapper { display: grid; grid-gap: 10px; @@ -186,9 +216,6 @@ summary > h3 { justify-content: space-between; } -.wrapper > div { -} - .org_title { grid-column: 1 / 3; grid-row: 1; @@ -218,6 +245,11 @@ summary > h3 { grid-row: 2; } +.org_top { + grid-column: 1/3; + grid-row: 3; +} + /* Vue marché */ .marche_infos p { diff --git a/src/callbacks.py b/src/callbacks.py new file mode 100644 index 0000000..6bf197e --- /dev/null +++ b/src/callbacks.py @@ -0,0 +1,65 @@ +import polars as pl +from dash import dash_table, html + +from utils import add_links_in_dict, format_values, setup_table_columns + + +def get_top_org_table(data, org_type: str): + dff = pl.DataFrame(data) + if dff.height == 0: + return html.Div() + + dff = dff.select( + ["uid", f"{org_type}_id", f"{org_type}_nom", "distance", "montant"] + ) + dff_nb = dff.group_by(f"{org_type}_id", f"{org_type}_nom", "distance").agg( + pl.len().alias("Attributions"), pl.sum("montant").alias("montant") + ) + dff_nb = dff_nb.sort(by="montant", descending=True) + dff_nb = dff_nb.cast(pl.String) + dff_nb = dff_nb.fill_null("") + dff_nb = format_values(dff_nb) + columns, tooltip = setup_table_columns( + dff_nb, hideable=False, exclude=[f"{org_type}_id"] + ) + data = dff_nb.to_dicts() + data = add_links_in_dict(data, f"{org_type}") + + return dash_table.DataTable( + data=data, + markdown_options={"html": True}, + page_action="native", + page_size=10, + columns=columns, + cell_selectable=False, + tooltip_header=tooltip, + style_cell_conditional=[ + { + "if": {"column_id": "objet"}, + "minWidth": "350px", + "textAlign": "left", + "overflow": "hidden", + "lineHeight": "14px", + "whiteSpace": "normal", + "fontSize": "85%", + }, + { + "if": {"column_id": "acheteur_nom"}, + "minWidth": "200px", + "textAlign": "left", + "overflow": "hidden", + "lineHeight": "16px", + # "fontSize": "85%", + "whiteSpace": "normal", + }, + { + "if": {"column_id": "titulaire_nom"}, + "minWidth": "200px", + "textAlign": "left", + "overflow": "hidden", + "lineHeight": "16px", + "whiteSpace": "normal", + # "fontSize": "85%", + }, + ], + ) diff --git a/src/pages/acheteur.py b/src/pages/acheteur.py index 80a258f..87ddd05 100644 --- a/src/pages/acheteur.py +++ b/src/pages/acheteur.py @@ -3,12 +3,13 @@ import datetime import polars as pl from dash import Input, Output, State, callback, dash_table, dcc, html, register_page +from src.callbacks import get_top_org_table from src.figures import point_on_map from src.utils import ( add_links_in_dict, df, - format_montant, format_number, + format_values, get_annuaire_data, get_departement_region, meta_content, @@ -89,6 +90,13 @@ layout = [ ], ), html.Div(className="org_map", id="acheteur_map"), + html.Div( + className="org_top", + children=[ + html.H3("Top titulaires"), + html.Div(className="marches_table", id="top10_titulaires"), + ], + ), ], ), # récupérer les données de l'acheteur sur l'api annuaire @@ -146,22 +154,22 @@ def update_acheteur_infos(url): Input(component_id="acheteur_data", component_property="data"), ) def update_acheteur_stats(data): - df = pl.DataFrame(data) - if df.height == 0: - df = pl.DataFrame(schema=df.collect_schema()) - df_marches = df.unique("id") + dff = pl.DataFrame(data) + if dff.height == 0: + dff = pl.DataFrame(schema=df.collect_schema()) + df_marches = dff.unique("id") nb_marches = format_number(df_marches.height) # somme_marches = format_number(int(df_marches.select(pl.sum("montant")).item())) marches_attribues = [html.Strong(nb_marches), " marchés et accord-cadres attribués"] # + ", pour un total de ", html.Strong(somme_marches + " €")] del df_marches - nb_titulaires = df.unique("titulaire_id").height + nb_titulaires = dff.unique("titulaire_id").height nb_titulaires = [ html.Strong(format_number(nb_titulaires)), " titulaires (SIRET) différents", ] - del df + del dff return marches_attribues, nb_titulaires @@ -183,6 +191,7 @@ def get_acheteur_marches_data(url, acheteur_year: str) -> list[dict]: "titulaire_id", "titulaire_typeIdentifiant", "titulaire_nom", + "distance", "montant", "codeCPV", "dureeMois", @@ -203,9 +212,11 @@ def get_acheteur_marches_data(url, acheteur_year: str) -> list[dict]: ) def get_last_marches_table(data) -> html.Div: dff = pl.DataFrame(data) + if dff.height == 0: + return html.Div(html.P("Aucun marché trouvé.")) dff = dff.cast(pl.String) dff = dff.fill_null("") - dff = format_montant(dff) + dff = format_values(dff) columns, tooltip = setup_table_columns( dff, hideable=False, @@ -235,7 +246,7 @@ def get_last_marches_table(data) -> html.Div: "minWidth": "300px", "textAlign": "left", "overflow": "hidden", - "lineHeight": "14px", + "lineHeight": "18px", "whiteSpace": "normal", }, { @@ -252,6 +263,14 @@ def get_last_marches_table(data) -> html.Div: return table +@callback( + Output(component_id="top10_titulaires", component_property="children"), + Input(component_id="acheteur_data", component_property="data"), +) +def get_top_titulaires(data): + return get_top_org_table(data, "titulaire") + + @callback( Output("download-acheteur-data", "data"), Input("btn-download-acheteur-data", "n_clicks"), diff --git a/src/pages/marche.py b/src/pages/marche.py index bd90a50..807bb7a 100644 --- a/src/pages/marche.py +++ b/src/pages/marche.py @@ -4,7 +4,7 @@ import polars as pl from dash import Input, Output, callback, dcc, html, register_page from polars import selectors as cs -from src.utils import data_schema, df, format_montant, meta_content +from src.utils import data_schema, df, format_values, meta_content register_page( __name__, @@ -75,7 +75,7 @@ def get_marche_data(url) -> tuple[dict, list]: # Données du marché dff_marche = lff.unique("uid").collect(engine="streaming") - dff_marche = format_montant(dff_marche) + dff_marche = format_values(dff_marche) return dff_marche.to_dicts()[0], dff_titulaires.to_dicts() diff --git a/src/pages/recherche.py b/src/pages/recherche.py new file mode 100644 index 0000000..b8b4796 --- /dev/null +++ b/src/pages/recherche.py @@ -0,0 +1,105 @@ +from dash import Input, Output, callback, dash_table, dcc, html, register_page + +from src.utils import ( + df_acheteurs, + df_titulaires, + meta_content, + search_org, + setup_table_columns, +) + +name = "Recherche" + +register_page( + __name__, + path="/", + title=meta_content["title"], + name=name, + description=meta_content["description"], + image_url=meta_content["image_url"], + order=0, +) + +layout = html.Div( + className="container", + children=[ + dcc.Input( + id="search", + type="text", + placeholder="Nom d'acheteur, d'entreprise, SIREN...", + autoFocus=True, + ), + # html.Div( + # className="search_options", + # children=[dcc.RadioItems(options=["Acheteur(s)"])], + # ), + html.Div(id="search_results", className="wrapper"), + ], +) + + +@callback( + Output("search_results", "children"), + Input("search", "value"), + prevent_initial_call=True, +) +def update_search_results(query): + if len(query) >= 1: + content = [] + + for org_type in ["acheteur", "titulaire"]: + if org_type == "acheteur": + dff = df_acheteurs + elif org_type == "titulaire": + dff = df_titulaires + else: + raise ValueError(f"{org_type} is not supported") + + # Search acheteurs and titulaires using the same function + results = search_org(dff, query, org_type=org_type) + count = results.height + + # Format output + columns, tooltip = setup_table_columns(results, hideable=False) + + org_content = [ + html.Div( + className=f"results_{org_type}", + children=[ + html.H3(f"{org_type.title()}s : {count}"), + dash_table.DataTable( + columns=columns, + data=results.to_dicts(), + page_size=10, + # style_table={"overflowX": "auto"}, + markdown_options={"html": True}, + cell_selectable=False, + style_cell_conditional=[ + { + "if": {"column_id": "acheteur_nom"}, + "maxWidth": "250px", + "textAlign": "left", + "overflow": "hidden", + "lineHeight": "18px", + "whiteSpace": "normal", + }, + { + "if": {"column_id": "titulaire_nom"}, + "maxWidth": "250px", + "textAlign": "left", + "overflow": "hidden", + "lineHeight": "18px", + "whiteSpace": "normal", + }, + ], + ), + ], + ) + if count > 0 + else html.P(f"Aucun {org_type} trouvé."), + ] + content.extend(org_content) + + return content + else: + return html.P("") diff --git a/src/pages/tableau.py b/src/pages/tableau.py index 23de881..972a240 100644 --- a/src/pages/tableau.py +++ b/src/pages/tableau.py @@ -9,8 +9,9 @@ from src.utils import ( add_resource_link, df, filter_table_data, - format_montant, format_number, + format_values, + get_default_hidden_columns, meta_content, setup_table_columns, sort_table_data, @@ -24,7 +25,7 @@ schema = df.collect_schema() name = "Tableau" register_page( __name__, - path="/", + path="/tableau", title=meta_content["title"], name=name, description=meta_content["description"], @@ -52,7 +53,7 @@ datatable = html.Div( "minWidth": "350px", "textAlign": "left", "overflow": "hidden", - "lineHeight": "14px", + "lineHeight": "18px", "whiteSpace": "normal", }, { @@ -60,7 +61,7 @@ datatable = html.Div( "minWidth": "250px", "textAlign": "left", "overflow": "hidden", - "lineHeight": "14px", + "lineHeight": "18px", "whiteSpace": "normal", }, { @@ -68,7 +69,7 @@ datatable = html.Div( "minWidth": "250px", "textAlign": "left", "overflow": "hidden", - "lineHeight": "14px", + "lineHeight": "18px", "whiteSpace": "normal", }, ], @@ -76,6 +77,7 @@ datatable = html.Div( markdown_options={"html": True}, tooltip_duration=8000, tooltip_delay=350, + hidden_columns=get_default_hidden_columns(schema), ), ) @@ -206,7 +208,7 @@ def update_table(page_current, page_size, filter_query, sort_by, data_timestamp) dff = add_resource_link(dff) # Formatage des montants - dff = format_montant(dff) + dff = format_values(dff) columns, tooltip = setup_table_columns(dff) diff --git a/src/pages/titulaire.py b/src/pages/titulaire.py index 07f07eb..3ebeaef 100644 --- a/src/pages/titulaire.py +++ b/src/pages/titulaire.py @@ -3,12 +3,13 @@ import datetime import polars as pl from dash import Input, Output, State, callback, dash_table, dcc, html, register_page +from src.callbacks import get_top_org_table from src.figures import point_on_map from src.utils import ( add_links_in_dict, df, - format_montant, format_number, + format_values, get_annuaire_data, get_departement_region, meta_content, @@ -91,6 +92,13 @@ layout = [ ], ), html.Div(className="org_map", id="titulaire_map"), + html.Div( + className="org_top", + children=[ + html.H3("Top acheteurs"), + html.Div(className="marches_table", id="top10_acheteurs"), + ], + ), ], ), # récupérer les données de l'acheteur sur l'api annuaire @@ -161,7 +169,7 @@ def update_titulaire_stats(data): nb_acheteurs = dff.unique("acheteur_id").height nb_acheteurs = [ html.Strong(format_number(nb_acheteurs)), - " titulaires (SIRET) différents", + " acheteurs (SIRET) différents", ] del dff @@ -187,6 +195,7 @@ def get_titulaire_marches_data(url, titulaire_year: str) -> list[dict]: "dateNotification", "acheteur_id", "acheteur_nom", + "distance", "montant", "codeCPV", "dureeMois", @@ -219,7 +228,7 @@ def get_last_marches_table(data) -> html.Div: dff = pl.DataFrame(data) dff = dff.cast(pl.String) dff = dff.fill_null("") - dff = format_montant(dff) + dff = format_values(dff) columns, tooltip = setup_table_columns( dff, hideable=False, exclude=["acheteur_id", "id"] ) @@ -249,7 +258,7 @@ def get_last_marches_table(data) -> html.Div: "minWidth": "300px", "textAlign": "left", "overflow": "hidden", - "lineHeight": "14px", + "lineHeight": "18px", "whiteSpace": "normal", }, { @@ -266,6 +275,14 @@ def get_last_marches_table(data) -> html.Div: return table +@callback( + Output(component_id="top10_acheteurs", component_property="children"), + Input(component_id="titulaire_data", component_property="data"), +) +def get_top_acheteurs(data): + return get_top_org_table(data, "acheteur") + + @callback( Output("download-titulaire-data", "data"), Input("btn-download-titulaire-data", "n_clicks"), diff --git a/src/utils.py b/src/utils.py index 85d9e43..eaaf9a5 100644 --- a/src/utils.py +++ b/src/utils.py @@ -1,22 +1,19 @@ import json import logging import os -from time import sleep +import uuid +from time import localtime, sleep import polars as pl import polars.selectors as cs -from httpx import get +from httpx import get, post +from polars import Schema from polars.exceptions import ComputeError - -operators = [ - ["s<", "<"], - ["s>", ">"], - ["i<", "<"], - ["i>", ">"], - ["icontains", "contains"], -] +from unidecode import unidecode logger = logging.getLogger("decp.info") +logging.getLogger("httpx").setLevel("WARNING") + logging.basicConfig( format="%(asctime)s %(levelname)-8s %(message)s", level=logging.INFO, @@ -25,6 +22,13 @@ logging.basicConfig( def split_filter_part(filter_part): + operators = [ + ["s<", "<"], + ["s>", ">"], + ["i<", "<"], + ["i>", ">"], + ["icontains", "contains"], + ] print("filter part", filter_part) for operator_group in operators: if operator_group[0] in filter_part: @@ -49,42 +53,60 @@ def add_resource_link(dff: pl.DataFrame) -> pl.DataFrame: return dff -def add_links(dff: pl.DataFrame): - dff = dff.with_columns( - pl.when(pl.col("titulaire_typeIdentifiant") == "SIRET") - .then( - '' - + pl.col("titulaire_id") - + "" - ) - .otherwise(pl.col("titulaire_id")) - .alias("titulaire_id") - ) - - for column, path in [("acheteur_id", "acheteurs"), ("uid", "marches")]: - dff = dff.with_columns( - ( - f'' - + pl.col(column) - + "" - ).alias(column) - ) +def add_links(dff: pl.DataFrame, target: str = "_blank"): + for col in ["uid", "acheteur_nom", "titulaire_nom", "acheteur_id", "titulaire_id"]: + if col in dff.columns: + if col.startswith("titulaire_"): + dff = dff.with_columns( + pl.when( + pl.Expr.or_( + pl.col("titulaire_typeIdentifiant").is_null(), + pl.col("titulaire_typeIdentifiant") == "SIRET", + ) + ) + .then( + '' + + pl.col(col) + + "" + ) + .otherwise(pl.col(col)) + .alias(col) + ) + if col.startswith("acheteur_"): + dff = dff.with_columns( + ( + '' + + pl.col(col) + + "" + ).alias(col) + ) + if col == "uid": + dff = dff.with_columns( + ( + '' + + pl.col("uid") + + "" + ).alias("uid") + ) return dff -def add_links_in_dict(data: list, org_type: str) -> list: +def add_links_in_dict(data: list[dict], org_type: str) -> list: new_data = [] for marche in data: org_id = marche[org_type + "_id"] marche[org_type + "_nom"] = ( f'{marche[org_type + "_nom"]}' ) - marche["id"] = f'{marche["id"]}' - marche["uid"] = f'{marche["uid"]}' + if marche.get("uid"): + marche["id"] = f'{marche["id"]}' + marche["uid"] = f'{marche["uid"]}' new_data.append(marche) return new_data @@ -123,8 +145,8 @@ def format_number(number) -> str: return number -def format_montant(dff: pl.DataFrame) -> pl.DataFrame: - def format_function(expr, scale=None): +def format_values(dff: pl.DataFrame) -> pl.DataFrame: + def format_montant(expr, scale=None): # https://stackoverflow.com/a/78636786 expr = expr.cast(pl.String) expr = expr.str.splitn(".", 2) @@ -142,7 +164,7 @@ def format_montant(dff: pl.DataFrame) -> pl.DataFrame: frac: pl.Expr = ( pl.when(frac.is_not_null() & ~frac.is_in(["0"])) - .then("," + frac) + .then("," + frac.str.head(2)) .otherwise(pl.lit("")) ) @@ -154,7 +176,17 @@ def format_montant(dff: pl.DataFrame) -> pl.DataFrame: return montant - dff = dff.with_columns(pl.col("montant").pipe(format_function).alias("montant")) + def format_distance(expr): + expr = expr.cast(pl.String) + return pl.concat_str(expr, pl.lit(" km")) + + if "montant" in dff.columns: + dff = dff.with_columns(pl.col("montant").pipe(format_montant).alias("montant")) + if "distance" in dff.columns: + dff = dff.with_columns( + pl.col("distance").pipe(format_distance).alias("distance") + ) + return dff @@ -196,6 +228,18 @@ def get_decp_data() -> pl.DataFrame: return lff.collect() +def get_org_data(dff: pl.DataFrame, org_type: str) -> pl.DataFrame: + lff = dff.lazy() + lff = lff.select( + "uid", + cs.starts_with(org_type).exclude( + f"{org_type}_latitude", f"{org_type}_longitude" + ), + ) + lff = lff.group_by(cs.starts_with(org_type)).len("Marchés") + return lff.collect() + + def get_departements() -> dict: with open("data/departements.json", "rb") as f: data = json.load(f) @@ -312,6 +356,20 @@ def setup_table_columns(dff, hideable: bool = True, exclude: list = None) -> tup return columns, tooltip +def get_default_hidden_columns(schema: Schema): + displayed_columns = os.getenv("DISPLAYED_COLUMNS") + hidden_columns = [] + if displayed_columns: + displayed_columns = displayed_columns.replace(" ", "").split(",") + for col in schema.names(): + if col in displayed_columns: + continue + else: + hidden_columns.append(col) + return hidden_columns + raise ValueError("DISPLAYED_COLUMNS n'est pas configuré") + + def get_data_schema() -> dict: # Récupération du schéma des données tabulaires path = os.getenv("DATA_SCHEMA_PATH") @@ -338,7 +396,114 @@ def get_data_schema() -> dict: return new_schema +def track_search(query): + if ( + len(query) >= 4 + and os.getenv("DEVELOPMENT").lower != "true" + and os.getenv("MATOMO_DOMAIN") + ): + if os.getenv("DEVELOPMENT").lower() == "true": + url = "https://test.decp.info" + else: + url = "https://decp.info" + params = { + "idsite": os.getenv("MATOMO_ID_SITE"), + "url": url, + "rec": "1", + "action_name": "front_page_search", + "rand": uuid.uuid4().hex, + "apiv": "1", + "h": localtime().tm_hour, + "m": localtime().tm_min, + "s": localtime().tm_sec, + "search": query, + "token_auth": os.getenv("MATOMO_TOKEN"), + } + post( + url=f"https://{os.getenv('MATOMO_DOMAIN')}/matomo.php", + params=params, + ).raise_for_status() + + +def search_org(dff: pl.DataFrame, query: str, org_type: str) -> pl.DataFrame: + """ + Search in either 'acheteur' or 'titulaire' DataFrame. + + :param dff: Polars DataFrame with acheteur or titulaire columns + :param query: User search string + :param org_type: 'acheteur' or 'titulaire' + :return: Filtered DataFrame with 'matches' column + """ + if not query.strip(): + return dff.select(pl.lit(False).alias("matches")) + + # Enregistrement des recherche dans Matomo + track_search(query) + + # Normalize query + normalized_query = unidecode(query.strip()).upper() + tokens = [" " + t.strip() for t in normalized_query.split() if t.strip()] + + # Define columns based on entity type + cols = [ + f"{org_type}_id", + f"{org_type}_nom", + f"{org_type}_departement_nom", + f"{org_type}_departement_code", + f"{org_type}_commune_nom", + ] + + # Concatenate all fields into one string per row + org_str = pl.concat_str(pl.lit(" "), pl.col(cols), separator=" ") + + # For each token, create a boolean column: True if token is found + token_matches = [] + for token in tokens: + token_match = org_str.str.contains(token).alias(f"token_{token}") + token_matches.append(token_match) + + # Count how many tokens match per row + match_score = pl.sum_horizontal(token_matches).alias("match_score") + + # For each token, create a boolean column: True if token is found + token_matches = [] + for token in tokens: + token_match = org_str.str.contains(token).alias(f"token_{token}") + token_matches.append(token_match) + + # Sélection des colonnes + if org_type == "acheteur": + dff = dff.select(cols + ["Marchés"]) + if org_type == "titulaire": + dff = dff.select(cols + ["Marchés", "titulaire_typeIdentifiant"]) + + # Apply and filter + dff = ( + dff.with_columns(token_matches + [match_score]) + .filter(pl.col("match_score") == len(tokens)) + .sort("Marchés", descending=True) + .drop([f"token_{token}" for token in tokens]) + ) + + # Format result + dff = add_links(dff, target="") + dff = dff.with_columns( + pl.concat_str( + pl.col(f"{org_type}_departement_nom"), + pl.lit(" ("), + pl.col(f"{org_type}_departement_code"), + pl.lit(")"), + ).alias("Département") + ) + + dff = dff.select(f"{org_type}_id", f"{org_type}_nom", "Département", "Marchés") + + return dff + + df: pl.DataFrame = get_decp_data() +df_acheteurs = get_org_data(df, "acheteur") +df_titulaires = get_org_data(df, "titulaire") departements = get_departements() domain_name = ( "test.decp.info" if os.getenv("DEVELOPMENT").lower() == "true" else "decp.info"