armanddemasson commited on
Commit
4076012
·
2 Parent(s): b21471ac9c6431

Merged in feature/talk_to_ipcc (pull request #23)

Browse files
app.py CHANGED
@@ -17,6 +17,7 @@ from front.tabs import create_config_modal, cqa_tab, create_about_tab
17
  from front.tabs import MainTabPanel, ConfigPanel
18
  from front.tabs.tab_drias import create_drias_tab
19
  from front.tabs.tab_ipcc import create_ipcc_tab
 
20
  from front.utils import process_figures
21
  from gradio_modal import Modal
22
 
@@ -535,6 +536,7 @@ def main_ui():
535
  local_cqa_components = cqa_tab(tab_name="France - Local Q&A")
536
  drias_components = create_drias_tab(share_client=share_client, user_id=user_id)
537
  ipcc_components = create_ipcc_tab(share_client=share_client, user_id=user_id)
 
538
  create_about_tab()
539
 
540
  event_handling(cqa_components, config_components, tab_name="ClimateQ&A")
 
17
  from front.tabs import MainTabPanel, ConfigPanel
18
  from front.tabs.tab_drias import create_drias_tab
19
  from front.tabs.tab_ipcc import create_ipcc_tab
20
+
21
  from front.utils import process_figures
22
  from gradio_modal import Modal
23
 
 
536
  local_cqa_components = cqa_tab(tab_name="France - Local Q&A")
537
  drias_components = create_drias_tab(share_client=share_client, user_id=user_id)
538
  ipcc_components = create_ipcc_tab(share_client=share_client, user_id=user_id)
539
+
540
  create_about_tab()
541
 
542
  event_handling(cqa_components, config_components, tab_name="ClimateQ&A")
climateqa/engine/talk_to_data/input_processing.py CHANGED
@@ -10,6 +10,7 @@ from climateqa.engine.talk_to_data.objects.llm_outputs import ArrayOutput
10
  from climateqa.engine.talk_to_data.objects.location import Location
11
  from climateqa.engine.talk_to_data.objects.plot import Plot
12
  from climateqa.engine.talk_to_data.objects.states import State
 
13
 
14
  async def detect_location_with_openai(sentence: str) -> str:
15
  """
@@ -46,7 +47,7 @@ def loc_to_coords(location: str) -> tuple[float, float]:
46
  Raises:
47
  AttributeError: If the location cannot be found
48
  """
49
- geolocator = Nominatim(user_agent="city_to_latlong")
50
  coords = geolocator.geocode(location)
51
  return (coords.latitude, coords.longitude)
52
 
@@ -118,7 +119,37 @@ async def detect_year_with_openai(sentence: str) -> str:
118
  return years_list[0]
119
  else:
120
  return ""
121
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
122
 
123
  async def detect_relevant_tables(user_question: str, plot: Plot, llm, table_names_list: list[str]) -> list[str]:
124
  """Identifies relevant tables for a plot based on user input.
@@ -227,6 +258,13 @@ async def find_year(user_input: str) -> str| None:
227
  return None
228
  return year
229
 
 
 
 
 
 
 
 
230
  async def find_relevant_plots(state: State, llm, plots: list[Plot]) -> list[str]:
231
  print("---- Find relevant plots ----")
232
  relevant_plots = await detect_relevant_plots(state['user_input'], llm, plots)
@@ -237,16 +275,9 @@ async def find_relevant_tables_per_plot(state: State, plot: Plot, llm, tables: l
237
  relevant_tables = await detect_relevant_tables(state['user_input'], plot, llm, tables)
238
  return relevant_tables
239
 
240
- async def find_param(state: State, param_name:str, mode: Literal['DRIAS', 'IPCC'] = 'DRIAS') -> dict[str, Optional[str]] | Location | None:
241
- """Perform the good method to retrieve the desired parameter
242
-
243
- Args:
244
- state (State): state of the workflow
245
- param_name (str): name of the desired parameter
246
- table (str): name of the table
247
-
248
- Returns:
249
- dict[str, Any] | None:
250
  """
251
  if param_name == 'location':
252
  location = await find_location(state['user_input'], mode)
@@ -254,4 +285,7 @@ async def find_param(state: State, param_name:str, mode: Literal['DRIAS', 'IPCC'
254
  if param_name == 'year':
255
  year = await find_year(state['user_input'])
256
  return {'year': year}
257
- return None
 
 
 
 
10
  from climateqa.engine.talk_to_data.objects.location import Location
11
  from climateqa.engine.talk_to_data.objects.plot import Plot
12
  from climateqa.engine.talk_to_data.objects.states import State
13
+ import calendar
14
 
15
  async def detect_location_with_openai(sentence: str) -> str:
16
  """
 
47
  Raises:
48
  AttributeError: If the location cannot be found
49
  """
50
+ geolocator = Nominatim(user_agent="city_to_latlong", timeout=5)
51
  coords = geolocator.geocode(location)
52
  return (coords.latitude, coords.longitude)
53
 
 
119
  return years_list[0]
120
  else:
121
  return ""
122
+
123
+ async def detect_month_with_openai(sentence: str) -> dict[str, str]:
124
+ """
125
+ Detects month in a sentence using OpenAI's API via LangChain.
126
+ Returns the month as an integer string (e.g., "7" for July), or "" if not found.
127
+ """
128
+ llm = get_llm()
129
+ prompt = """
130
+ Extract the month (as a number from 1 to 12) mentioned in the following sentence.
131
+ Return the result as a Python list of integers. If no month is mentioned, return an empty list.
132
+
133
+ Sentence: "{sentence}"
134
+ """
135
+ prompt = ChatPromptTemplate.from_template(prompt)
136
+ structured_llm = llm.with_structured_output(ArrayOutput)
137
+ chain = prompt | structured_llm
138
+ response: ArrayOutput = await chain.ainvoke({"sentence": sentence})
139
+ months_list = eval(response['array'])
140
+ if len(months_list) > 0:
141
+ month_number = int(months_list[0])
142
+ month_name = calendar.month_name[month_number]
143
+ return {
144
+ "month_number": str(month_number),
145
+ "month_name": month_name
146
+ }
147
+ else:
148
+ return {
149
+ "month_number" : "",
150
+ "month_name" : ""
151
+ }
152
+
153
 
154
  async def detect_relevant_tables(user_question: str, plot: Plot, llm, table_names_list: list[str]) -> list[str]:
155
  """Identifies relevant tables for a plot based on user input.
 
258
  return None
259
  return year
260
 
261
+ async def find_month(user_input: str) -> dict[str, str|None]:
262
+ """Extracts month information from user input using LLM."""
263
+ print(f"---- Find month ---")
264
+ month_info = await detect_month_with_openai(user_input)
265
+ month_info = {key: None if value == "" else value for key, value in month_info.items()}
266
+ return month_info
267
+
268
  async def find_relevant_plots(state: State, llm, plots: list[Plot]) -> list[str]:
269
  print("---- Find relevant plots ----")
270
  relevant_plots = await detect_relevant_plots(state['user_input'], llm, plots)
 
275
  relevant_tables = await detect_relevant_tables(state['user_input'], plot, llm, tables)
276
  return relevant_tables
277
 
278
+ async def find_param(state: State, param_name: str, mode: Literal['DRIAS', 'IPCC'] = 'DRIAS') -> dict[str, Optional[str]] | Location | None:
279
+ """
280
+ Perform the good method to retrieve the desired parameter.
 
 
 
 
 
 
 
281
  """
282
  if param_name == 'location':
283
  location = await find_location(state['user_input'], mode)
 
285
  if param_name == 'year':
286
  year = await find_year(state['user_input'])
287
  return {'year': year}
288
+ if param_name == 'month':
289
+ month = await find_month(state['user_input'])
290
+ return month
291
+ return None
climateqa/engine/talk_to_data/ipcc/config.py CHANGED
@@ -6,16 +6,22 @@ from climateqa.engine.talk_to_data.config import IPCC_DATASET_URL
6
  IPCC_TABLES = [
7
  "mean_temperature",
8
  "total_precipitation",
 
 
9
  ]
10
 
11
  IPCC_INDICATOR_COLUMNS_PER_TABLE = {
12
  "mean_temperature": "mean_temperature",
13
- "total_precipitation": "total_precipitation"
 
 
14
  }
15
 
16
  IPCC_INDICATOR_TO_UNIT = {
17
  "mean_temperature": "°C",
18
- "total_precipitation": "mm/day"
 
 
19
  }
20
 
21
  IPCC_SCENARIO = [
@@ -30,7 +36,8 @@ IPCC_MODELS = []
30
 
31
  IPCC_PLOT_PARAMETERS = [
32
  'year',
33
- 'location'
 
34
  ]
35
 
36
  MACRO_COUNTRIES = ['JP',
@@ -63,7 +70,9 @@ HUGE_MACRO_COUNTRIES = ['CL',
63
 
64
  IPCC_INDICATOR_TO_COLORSCALE = {
65
  "mean_temperature": TEMPERATURE_COLORSCALE,
66
- "total_precipitation": PRECIPITATION_COLORSCALE
 
 
67
  }
68
 
69
  IPCC_UI_TEXT = """
@@ -77,9 +86,12 @@ By default, we take the **mediane of each climate model**.
77
  Current available charts :
78
  - Yearly evolution of an indicator at a specific location (historical + SSP Projections)
79
  - Yearly spatial distribution of an indicator in a specific country
 
80
 
81
  Current available indicators :
82
  - Mean temperature
 
 
83
  - Total precipitation
84
 
85
  For example, you can ask:
 
6
  IPCC_TABLES = [
7
  "mean_temperature",
8
  "total_precipitation",
9
+ "minimum_temperature",
10
+ "maximum_temperature"
11
  ]
12
 
13
  IPCC_INDICATOR_COLUMNS_PER_TABLE = {
14
  "mean_temperature": "mean_temperature",
15
+ "total_precipitation": "total_precipitation",
16
+ "minimum_temperature": "minimum_temperature",
17
+ "maximum_temperature": "maximum_temperature"
18
  }
19
 
20
  IPCC_INDICATOR_TO_UNIT = {
21
  "mean_temperature": "°C",
22
+ "total_precipitation": "mm/day",
23
+ "minimum_temperature": "°C",
24
+ "maximum_temperature": "°C"
25
  }
26
 
27
  IPCC_SCENARIO = [
 
36
 
37
  IPCC_PLOT_PARAMETERS = [
38
  'year',
39
+ 'location',
40
+ 'month'
41
  ]
42
 
43
  MACRO_COUNTRIES = ['JP',
 
70
 
71
  IPCC_INDICATOR_TO_COLORSCALE = {
72
  "mean_temperature": TEMPERATURE_COLORSCALE,
73
+ "total_precipitation": PRECIPITATION_COLORSCALE,
74
+ "minimum_temperature": TEMPERATURE_COLORSCALE,
75
+ "maximum_temperature": TEMPERATURE_COLORSCALE,
76
  }
77
 
78
  IPCC_UI_TEXT = """
 
86
  Current available charts :
87
  - Yearly evolution of an indicator at a specific location (historical + SSP Projections)
88
  - Yearly spatial distribution of an indicator in a specific country
89
+ - Yearly evolution of an indicator in a specific month at a specific location (historical + SSP Projections)
90
 
91
  Current available indicators :
92
  - Mean temperature
93
+ - Minimum temperature
94
+ - Maximum temperature
95
  - Total precipitation
96
 
97
  For example, you can ask:
climateqa/engine/talk_to_data/ipcc/plot_informations.py CHANGED
@@ -47,4 +47,27 @@ Each grid point is colored according to the value of the indicator ({unit}), all
47
  - For each grid point of {location} country ({country_name}), the value of {indicator} in {year} and for the selected scenario is extracted and mapped to its geographic coordinates.
48
  - The grid points correspond to 1-degree squares centered on the grid points of the IPCC dataset. Each grid point has been mapped to a country using [**reverse_geocoder**](https://github.com/thampiman/reverse-geocoder).
49
  - The coordinates used for each region are those of the closest available grid point in the IPCC database, which uses a regular grid with a spatial resolution of 1 degree.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
50
  """
 
47
  - For each grid point of {location} country ({country_name}), the value of {indicator} in {year} and for the selected scenario is extracted and mapped to its geographic coordinates.
48
  - The grid points correspond to 1-degree squares centered on the grid points of the IPCC dataset. Each grid point has been mapped to a country using [**reverse_geocoder**](https://github.com/thampiman/reverse-geocoder).
49
  - The coordinates used for each region are those of the closest available grid point in the IPCC database, which uses a regular grid with a spatial resolution of 1 degree.
50
+ """
51
+
52
+ def indicator_specific_month_evolution_informations(
53
+ indicator: str,
54
+ params: dict[str, str]
55
+ ) -> str:
56
+ if "location" not in params:
57
+ raise ValueError('"location" must be provided in params')
58
+ location = params["location"]
59
+ if "month_name" not in params:
60
+ raise ValueError('"month_name" must be provided in params')
61
+ month = params["month_name"]
62
+ unit = IPCC_INDICATOR_TO_UNIT[indicator]
63
+ return f"""
64
+ This plot shows how the climate indicator **{indicator}** evolves over time in **{location}** for the month of **{month}**.
65
+ It combines both historical (from 1950 to 2015) observations and future (from 2016 to 2100) projections for the different SSP climate scenarios (SSP126, SSP245, SSP370 and SSP585).
66
+ The x-axis represents the years (from 1950 to 2100), and the y-axis shows the value of the {indicator} ({unit}) for the selected month.
67
+ Each line corresponds to a different scenario, allowing you to compare how {indicator} for month {month} might change under various future conditions.
68
+
69
+ **Data source:**
70
+ - The data comes from the IPCC climate datasets (Parquet files) for the relevant indicator, location, and month.
71
+ - For each year and scenario, the value of {indicator} for month {month} is extracted for the selected location.
72
+ - The coordinates used for {location} correspond to the closest available point in the IPCC database, which uses a regular grid with a spatial resolution of 1 degree.
73
  """
climateqa/engine/talk_to_data/ipcc/plots.py CHANGED
@@ -5,8 +5,8 @@ import pandas as pd
5
  import geojson
6
 
7
  from climateqa.engine.talk_to_data.ipcc.config import IPCC_INDICATOR_TO_COLORSCALE, IPCC_INDICATOR_TO_UNIT, IPCC_SCENARIO
8
- from climateqa.engine.talk_to_data.ipcc.plot_informations import choropleth_map_informations, indicator_evolution_informations
9
- from climateqa.engine.talk_to_data.ipcc.queries import indicator_for_given_year_query, indicator_per_year_at_location_query
10
  from climateqa.engine.talk_to_data.objects.plot import Plot
11
 
12
  def generate_geojson_polygons(latitudes: list[float], longitudes: list[float], indicators: list[float]) -> geojson.FeatureCollection:
@@ -102,6 +102,82 @@ indicator_evolution_at_location_historical_and_projections: Plot = {
102
  "short_name": "Evolution"
103
  }
104
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
105
  def plot_choropleth_map_of_country_indicator_for_specific_year(
106
  params: dict,
107
  ) -> Callable[[pd.DataFrame], Figure]:
@@ -167,6 +243,7 @@ def plot_choropleth_map_of_country_indicator_for_specific_year(
167
 
168
  return plot_data
169
 
 
170
  choropleth_map_of_country_indicator_for_specific_year: Plot = {
171
  "name": "Choropleth Map of a Country's Indicator Distribution for a Specific Year",
172
  "description": (
@@ -185,5 +262,6 @@ choropleth_map_of_country_indicator_for_specific_year: Plot = {
185
 
186
  IPCC_PLOTS = [
187
  indicator_evolution_at_location_historical_and_projections,
188
- choropleth_map_of_country_indicator_for_specific_year
 
189
  ]
 
5
  import geojson
6
 
7
  from climateqa.engine.talk_to_data.ipcc.config import IPCC_INDICATOR_TO_COLORSCALE, IPCC_INDICATOR_TO_UNIT, IPCC_SCENARIO
8
+ from climateqa.engine.talk_to_data.ipcc.plot_informations import choropleth_map_informations, indicator_evolution_informations, indicator_specific_month_evolution_informations
9
+ from climateqa.engine.talk_to_data.ipcc.queries import indicator_for_given_year_query, indicator_per_year_and_specific_month_at_location_query, indicator_per_year_at_location_query
10
  from climateqa.engine.talk_to_data.objects.plot import Plot
11
 
12
  def generate_geojson_polygons(latitudes: list[float], longitudes: list[float], indicators: list[float]) -> geojson.FeatureCollection:
 
102
  "short_name": "Evolution"
103
  }
104
 
105
+ def plot_indicator_monthly_evolution_at_location(
106
+ params: dict,
107
+ ) -> Callable[[pd.DataFrame], Figure]:
108
+ """
109
+ Returns a function that generates a line plot showing the evolution of a climate indicator
110
+ for a specific month over time at a specific location, including both historical data
111
+ and future projections for different climate scenarios.
112
+
113
+ Args:
114
+ params (dict): Dictionary with:
115
+ - indicator_column (str): Name of the climate indicator column to plot.
116
+ - location (str): Location (e.g., country, city) for which to plot the indicator.
117
+ - month (str): Month name to plot.
118
+
119
+ Returns:
120
+ Callable[[pd.DataFrame], Figure]: Function that takes a DataFrame and returns a Plotly Figure.
121
+ """
122
+ indicator = params["indicator_column"]
123
+ location = params["location"]
124
+ month = params["month_name"]
125
+ indicator_label = " ".join(word.capitalize() for word in indicator.split("_"))
126
+ unit = IPCC_INDICATOR_TO_UNIT.get(indicator, "")
127
+
128
+ def plot_data(df: pd.DataFrame) -> Figure:
129
+ df = df.sort_values(by='year')
130
+ years = df['year'].astype(int).tolist()
131
+ indicators = df[indicator].astype(float).tolist()
132
+ scenarios = df['scenario'].astype(str).tolist()
133
+
134
+ # Find last historical value for continuity
135
+ last_historical = [(y, v) for y, v, s in zip(years, indicators, scenarios) if s == 'historical']
136
+ last_historical_year, last_historical_indicator = last_historical[-1] if last_historical else (None, None)
137
+
138
+ fig = go.Figure()
139
+ for scenario in IPCC_SCENARIO:
140
+ x = [y for y, s in zip(years, scenarios) if s == scenario]
141
+ y = [v for v, s in zip(indicators, scenarios) if s == scenario]
142
+ # Connect historical to scenario
143
+ if scenario != 'historical' and last_historical_indicator is not None:
144
+ x = [last_historical_year] + x
145
+ y = [last_historical_indicator] + y
146
+ fig.add_trace(go.Scatter(
147
+ x=x,
148
+ y=y,
149
+ mode='lines',
150
+ name=scenario
151
+ ))
152
+
153
+ fig.update_layout(
154
+ title=f'Evolution of {indicator_label} in {month} in {location} (Historical + SSP Scenarios)',
155
+ xaxis_title='Year',
156
+ yaxis_title=f'{indicator_label} ({unit})',
157
+ legend_title='Scenario',
158
+ height=800,
159
+ )
160
+ return fig
161
+
162
+ return plot_data
163
+
164
+
165
+ indicator_specific_month_evolution_at_location: Plot = {
166
+ "name": "Indicator specific month Evolution at Location (Historical + Projections)",
167
+ "description": (
168
+ "Shows how a climate indicator (e.g., rainfall, temperature) for a specific month changes over time at a specific location, "
169
+ "including historical data and future projections. "
170
+ "Useful for questions about the value or trend of an indicator for a given month at a location, "
171
+ "such as 'How does July temperature evolve in Paris over time?'. "
172
+ "Parameters: indicator_column (the climate variable), location (e.g., country, city), month (1-12)."
173
+ ),
174
+ "params": ["indicator_column", "location", "month"],
175
+ "plot_function": plot_indicator_monthly_evolution_at_location,
176
+ "sql_query": indicator_per_year_and_specific_month_at_location_query,
177
+ "plot_information": indicator_specific_month_evolution_informations,
178
+ "short_name": "Evolution for a specific month"
179
+ }
180
+
181
  def plot_choropleth_map_of_country_indicator_for_specific_year(
182
  params: dict,
183
  ) -> Callable[[pd.DataFrame], Figure]:
 
243
 
244
  return plot_data
245
 
246
+
247
  choropleth_map_of_country_indicator_for_specific_year: Plot = {
248
  "name": "Choropleth Map of a Country's Indicator Distribution for a Specific Year",
249
  "description": (
 
262
 
263
  IPCC_PLOTS = [
264
  indicator_evolution_at_location_historical_and_projections,
265
+ choropleth_map_of_country_indicator_for_specific_year,
266
+ indicator_specific_month_evolution_at_location
267
  ]
climateqa/engine/talk_to_data/ipcc/queries.py CHANGED
@@ -2,6 +2,7 @@ from typing import TypedDict, Optional
2
 
3
  from climateqa.engine.talk_to_data.ipcc.config import HUGE_MACRO_COUNTRIES, MACRO_COUNTRIES
4
  from climateqa.engine.talk_to_data.config import IPCC_DATASET_URL
 
5
  class IndicatorPerYearAtLocationQueryParams(TypedDict, total=False):
6
  """
7
  Parameters for querying the evolution of an indicator per year at a specific location.
@@ -42,7 +43,7 @@ def indicator_per_year_at_location_query(
42
  return ""
43
 
44
  if country_code in MACRO_COUNTRIES:
45
- table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_macro.parquet'"
46
  sql_query = f"""
47
  SELECT year, scenario, AVG({indicator_column}) as {indicator_column}
48
  FROM {table_path}
@@ -51,12 +52,12 @@ def indicator_per_year_at_location_query(
51
  ORDER BY year, scenario
52
  """
53
  elif country_code in HUGE_MACRO_COUNTRIES:
54
- table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_macro.parquet'"
55
  sql_query = f"""
56
- SELECT year, scenario, {indicator_column},
57
  FROM {table_path}
58
  WHERE latitude = {latitude} AND longitude = {longitude} AND year >= 1950
59
- ORDER year, scenario
60
  """
61
  else:
62
  table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}.parquet'"
@@ -74,6 +75,66 @@ def indicator_per_year_at_location_query(
74
  """
75
  return sql_query.strip()
76
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
  class IndicatorForGivenYearQueryParams(TypedDict, total=False):
78
  """
79
  Parameters for querying an indicator's values across locations for a specific year.
@@ -109,7 +170,7 @@ def indicator_for_given_year_query(
109
  return ""
110
 
111
  if country_code in MACRO_COUNTRIES:
112
- table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_macro.parquet'"
113
  sql_query = f"""
114
  SELECT latitude, longitude, scenario, AVG({indicator_column}) as {indicator_column}
115
  FROM {table_path}
@@ -118,9 +179,9 @@ def indicator_for_given_year_query(
118
  ORDER BY latitude, longitude, scenario
119
  """
120
  elif country_code in HUGE_MACRO_COUNTRIES:
121
- table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_macro.parquet'"
122
  sql_query = f"""
123
- SELECT latitude, longitude, scenario, {indicator_column},
124
  FROM {table_path}
125
  WHERE year = {year}
126
  ORDER BY latitude, longitude, scenario
@@ -140,4 +201,4 @@ def indicator_for_given_year_query(
140
  ORDER BY latitude, longitude, scenario
141
  """
142
 
143
- return sql_query.strip()
 
2
 
3
  from climateqa.engine.talk_to_data.ipcc.config import HUGE_MACRO_COUNTRIES, MACRO_COUNTRIES
4
  from climateqa.engine.talk_to_data.config import IPCC_DATASET_URL
5
+
6
  class IndicatorPerYearAtLocationQueryParams(TypedDict, total=False):
7
  """
8
  Parameters for querying the evolution of an indicator per year at a specific location.
 
43
  return ""
44
 
45
  if country_code in MACRO_COUNTRIES:
46
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_monthly_macro.parquet'"
47
  sql_query = f"""
48
  SELECT year, scenario, AVG({indicator_column}) as {indicator_column}
49
  FROM {table_path}
 
52
  ORDER BY year, scenario
53
  """
54
  elif country_code in HUGE_MACRO_COUNTRIES:
55
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_annualy_macro.parquet'"
56
  sql_query = f"""
57
+ SELECT year, scenario, {indicator_column}
58
  FROM {table_path}
59
  WHERE latitude = {latitude} AND longitude = {longitude} AND year >= 1950
60
+ ORDER BY year, scenario
61
  """
62
  else:
63
  table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}.parquet'"
 
75
  """
76
  return sql_query.strip()
77
 
78
+ class IndicatorPerYearAndSpecificMonthAtLocationQueryParams(TypedDict, total=False):
79
+ """
80
+ Parameters for querying the evolution of an indicator per year for a specific month at a specific location.
81
+
82
+ Attributes:
83
+ indicator_column (str): Name of the climate indicator column.
84
+ latitude (str): Latitude of the location.
85
+ longitude (str): Longitude of the location.
86
+ country_code (str): Country code.
87
+ month (str): Month targeted
88
+ """
89
+ indicator_column: str
90
+ latitude: str
91
+ longitude: str
92
+ country_code: str
93
+ month: str
94
+
95
+ def indicator_per_year_and_specific_month_at_location_query(
96
+ table: str, params: IndicatorPerYearAndSpecificMonthAtLocationQueryParams
97
+ ) -> str:
98
+ """
99
+ Builds an SQL query to get the evolution of an indicator per year for a specific month at a specific location.
100
+
101
+ Args:
102
+ table (str): SQL table of the indicator.
103
+ params (dict): Dictionary with required params:
104
+ - indicator_column (str)
105
+ - latitude (str or float)
106
+ - longitude (str or float)
107
+ - month (int)
108
+
109
+ Returns:
110
+ str: The SQL query string.
111
+ """
112
+ indicator_column = params.get("indicator_column")
113
+ latitude = params.get("latitude")
114
+ longitude = params.get("longitude")
115
+ country_code = params.get("country_code")
116
+ month = params.get('month_number')
117
+
118
+ if not all([indicator_column, latitude, longitude, country_code, month]):
119
+ return ""
120
+
121
+ if country_code in (MACRO_COUNTRIES+HUGE_MACRO_COUNTRIES):
122
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_monthly_macro.parquet'"
123
+ sql_query = f"""
124
+ SELECT year, scenario, {indicator_column}
125
+ FROM {table_path}
126
+ WHERE latitude = {latitude} AND longitude = {longitude} AND year >= 1950 AND month={month}
127
+ ORDER BY year, scenario
128
+ """
129
+ else:
130
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}.parquet'"
131
+ sql_query = f"""
132
+ SELECT year, scenario, MEDIAN({indicator_column}) AS {indicator_column}
133
+ FROM {table_path}
134
+ WHERE latitude = {latitude} AND longitude = {longitude} AND year >= 1950 AND month={month}
135
+ GROUP BY scenario, year
136
+ """
137
+ return sql_query.strip()
138
  class IndicatorForGivenYearQueryParams(TypedDict, total=False):
139
  """
140
  Parameters for querying an indicator's values across locations for a specific year.
 
170
  return ""
171
 
172
  if country_code in MACRO_COUNTRIES:
173
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_monthly_macro.parquet'"
174
  sql_query = f"""
175
  SELECT latitude, longitude, scenario, AVG({indicator_column}) as {indicator_column}
176
  FROM {table_path}
 
179
  ORDER BY latitude, longitude, scenario
180
  """
181
  elif country_code in HUGE_MACRO_COUNTRIES:
182
+ table_path = f"'{IPCC_DATASET_URL}/{table.lower()}/{country_code}_annualy_macro.parquet'"
183
  sql_query = f"""
184
+ SELECT latitude, longitude, scenario, {indicator_column}
185
  FROM {table_path}
186
  WHERE year = {year}
187
  ORDER BY latitude, longitude, scenario
 
201
  ORDER BY latitude, longitude, scenario
202
  """
203
 
204
+ return sql_query.strip()
climateqa/engine/talk_to_data/main.py CHANGED
@@ -50,7 +50,7 @@ async def ask_drias(query: str, index_state: int = 0, user_id: str | None = None
50
 
51
  if "error" in final_state and final_state["error"] != "":
52
  # No Sql query, no dataframe, no figure, no plot information, empty sql queries list, empty result dataframes list, empty figures list, empty plot information list, index state = 0, empty table list, error message
53
- return None, None, None, None, [], [], [], 0, [], final_state["error"]
54
 
55
  sql_query = sql_queries[index_state]
56
  dataframe = result_dataframes[index_state]
@@ -112,7 +112,7 @@ async def ask_ipcc(query: str, index_state: int = 0, user_id: str | None = None)
112
 
113
  if "error" in final_state and final_state["error"] != "":
114
  # No Sql query, no dataframe, no figure, no plot information, empty sql queries list, empty result dataframes list, empty figures list, empty plot information list, index state = 0, empty table list, error message
115
- return None, None, None, None, [], [], [], 0, [], final_state["error"]
116
 
117
  sql_query = sql_queries[index_state]
118
  dataframe = result_dataframes[index_state]
@@ -121,4 +121,4 @@ async def ask_ipcc(query: str, index_state: int = 0, user_id: str | None = None)
121
 
122
  log_drias_interaction_to_huggingface(query, sql_query, user_id)
123
 
124
- return sql_query, dataframe, figure, plot_information, sql_queries, result_dataframes, figures, plot_informations, index_state, plot_title_list, ""
 
50
 
51
  if "error" in final_state and final_state["error"] != "":
52
  # No Sql query, no dataframe, no figure, no plot information, empty sql queries list, empty result dataframes list, empty figures list, empty plot information list, index state = 0, empty table list, error message
53
+ return None, None, None, None, [], [], [], [], 0, [], final_state["error"]
54
 
55
  sql_query = sql_queries[index_state]
56
  dataframe = result_dataframes[index_state]
 
112
 
113
  if "error" in final_state and final_state["error"] != "":
114
  # No Sql query, no dataframe, no figure, no plot information, empty sql queries list, empty result dataframes list, empty figures list, empty plot information list, index state = 0, empty table list, error message
115
+ return None, None, None, None, [], [], [], [], 0, [], final_state["error"]
116
 
117
  sql_query = sql_queries[index_state]
118
  dataframe = result_dataframes[index_state]
 
121
 
122
  log_drias_interaction_to_huggingface(query, sql_query, user_id)
123
 
124
+ return sql_query, dataframe, figure, plot_information, sql_queries, result_dataframes, figures, plot_informations, index_state, plot_title_list, ""
climateqa/engine/talk_to_data/query.py CHANGED
@@ -3,6 +3,8 @@ from concurrent.futures import ThreadPoolExecutor
3
  import duckdb
4
  import pandas as pd
5
  import os
 
 
6
 
7
  def find_indicator_column(table: str, indicator_columns_per_table: dict[str,str]) -> str:
8
  """Retrieves the name of the indicator column within a table.
@@ -41,14 +43,88 @@ async def execute_sql_query(sql_query: str) -> pd.DataFrame:
41
  def _execute_query():
42
  # Execute the query
43
  con = duckdb.connect()
44
- HF_TOKEN = os.getenv("HF_TOKEN")
45
- con.execute(f"""CREATE SECRET hf_token (
46
- TYPE huggingface,
47
- TOKEN '{HF_TOKEN}'
48
- );""")
49
- results = con.execute(sql_query).fetchdf()
50
- # return fetched data
51
- return results
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
52
 
53
  # Run the query in a thread pool to avoid blocking
54
  loop = asyncio.get_event_loop()
 
3
  import duckdb
4
  import pandas as pd
5
  import os
6
+ import requests
7
+ import tempfile
8
 
9
  def find_indicator_column(table: str, indicator_columns_per_table: dict[str,str]) -> str:
10
  """Retrieves the name of the indicator column within a table.
 
43
  def _execute_query():
44
  # Execute the query
45
  con = duckdb.connect()
46
+
47
+ # Try to use Hugging Face authentication if token is available
48
+ HF_TTD_TOKEN = os.getenv("HF_TTD_TOKEN")
49
+
50
+ try:
51
+ if HF_TTD_TOKEN:
52
+ # Set up Hugging Face authentication - updated syntax
53
+ con.execute(f"""
54
+ CREATE SECRET IF NOT EXISTS hf_token (
55
+ TYPE HUGGINGFACE,
56
+ TOKEN '{HF_TTD_TOKEN}'
57
+ );
58
+ """)
59
+ print("Hugging Face authentication configured")
60
+
61
+ # Execute the query
62
+ results = con.execute(sql_query).fetchdf()
63
+ return results
64
+
65
+ except duckdb.HTTPException as e:
66
+ print(f"HTTP error accessing Hugging Face dataset: {e}")
67
+
68
+ # If we have a token but still get HTTP error, try without authentication
69
+ if HF_TTD_TOKEN:
70
+ print("Retrying without authentication...")
71
+ try:
72
+ # Create a new connection without the secret
73
+ con_no_auth = duckdb.connect()
74
+ results = con_no_auth.execute(sql_query).fetchdf()
75
+ return results
76
+ except Exception as e2:
77
+ print(f"Also failed without authentication: {e2}")
78
+
79
+ # Try to download the file locally and retry
80
+ print("Trying to download file locally and retry...")
81
+
82
+ # Extract the URL from the error message or construct it from the query
83
+ error_str = str(e)
84
+ url = None
85
+
86
+ if "HTTP GET error on '" in error_str:
87
+ url = error_str.split("HTTP GET error on '")[1].split("'")[0]
88
+ else:
89
+ # Try to extract URL from the SQL query
90
+ import re
91
+ url_match = re.search(r"'(https://huggingface\.co/[^']+)'", sql_query)
92
+ if url_match:
93
+ url = url_match.group(1)
94
+
95
+ if url:
96
+ table_name = url.split('/')[-1]
97
+ local_path = os.path.join(tempfile.gettempdir(), table_name)
98
+ print(f"Downloading {url} to {local_path}")
99
+
100
+ # Add authentication headers if token is available
101
+ headers = {}
102
+ if HF_TTD_TOKEN:
103
+ headers['Authorization'] = f'Bearer {HF_TTD_TOKEN}'
104
+
105
+ response = requests.get(url, headers=headers, stream=True)
106
+ if response.status_code == 200:
107
+ with open(local_path, 'wb') as f:
108
+ for chunk in response.iter_content(chunk_size=8192):
109
+ f.write(chunk)
110
+
111
+ # Modify the SQL query to use the local file
112
+ modified_sql = sql_query.replace(f"'{url}'", f"'{local_path}'")
113
+ results = con.execute(modified_sql).fetchdf()
114
+ return results
115
+ elif response.status_code == 401:
116
+ print("Authentication failed - check your HF_TTD_TOKEN")
117
+ raise Exception("Authentication failed. Please check your HF_TTD_TOKEN environment variable.")
118
+ else:
119
+ print(f"Failed to download file: {response.status_code}")
120
+ raise e
121
+ else:
122
+ print("Could not extract URL from error message")
123
+ raise e
124
+
125
+ except Exception as e:
126
+ print(f"Unexpected error: {e}")
127
+ raise e
128
 
129
  # Run the query in a thread pool to avoid blocking
130
  loop = asyncio.get_event_loop()
climateqa/engine/talk_to_data/workflow/ipcc.py CHANGED
@@ -152,10 +152,18 @@ async def ipcc_workflow(user_input: str) -> State:
152
 
153
  # Set error messages if needed
154
  if not errors['have_relevant_table']:
155
- state['error'] = "There is no relevant table in our database to answer your question"
 
 
 
156
  elif not errors['have_sql_query']:
157
- state['error'] = "There is no relevant sql query on our database that can help to answer your question"
 
 
 
158
  elif not errors['have_dataframe']:
159
- state['error'] = "There is no data in our table that can answer to your question"
160
-
 
 
161
  return state
 
152
 
153
  # Set error messages if needed
154
  if not errors['have_relevant_table']:
155
+ state['error'] = (
156
+ "Sorry, I couldn't find any relevant table in our database to answer your question.\n"
157
+ "Try asking about a different climate indicator like temperature or precipitation."
158
+ )
159
  elif not errors['have_sql_query']:
160
+ state['error'] = (
161
+ "Sorry, I couldn't generate a relevant SQL query to answer your question.\n"
162
+ "Try rephrasing your question to focus on a specific location, a year, or a month."
163
+ )
164
  elif not errors['have_dataframe']:
165
+ state['error'] = (
166
+ "Sorry, there is no data in our tables that can answer your question.\n"
167
+ "Try asking about a more common location, or a different year."
168
+ )
169
  return state
front/tabs/tab_ipcc.py CHANGED
@@ -68,6 +68,8 @@ def show_filter_by_scenario(table_names, index_state, dataframes):
68
  return gr.update(visible=False)
69
 
70
  def filter_by_scenario(dataframes, figures, table_names, index_state, scenario):
 
 
71
  df = dataframes[index_state]
72
  if not table_names[index_state].startswith("Map"):
73
  return df, figures[index_state](df)
 
68
  return gr.update(visible=False)
69
 
70
  def filter_by_scenario(dataframes, figures, table_names, index_state, scenario):
71
+ if len(dataframes) == 0:
72
+ return None, None
73
  df = dataframes[index_state]
74
  if not table_names[index_state].startswith("Map"):
75
  return df, figures[index_state](df)
requirements.txt CHANGED
@@ -26,4 +26,5 @@ duckdb==1.2.1
26
  openai==1.61.1
27
  pydantic==2.9.2
28
  pydantic-settings==2.2.1
29
- geojson==3.2.0
 
 
26
  openai==1.61.1
27
  pydantic==2.9.2
28
  pydantic-settings==2.2.1
29
+ geojson==3.2.0
30
+ requests==2.32.3
style.css CHANGED
@@ -741,4 +741,4 @@ div#tab-vanna{
741
  #example-img-container {
742
  flex-direction: column;
743
  align-items: left;
744
- }
 
741
  #example-img-container {
742
  flex-direction: column;
743
  align-items: left;
744
+ }