dahutapea commited on
Commit
7ab3f4b
·
1 Parent(s): 4eda25e

Expand corpus to 18 companies (recent SEC 10-Ks); ship prebuilt index via git-lfs

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ vectorstore/** filter=lfs diff=lfs merge=lfs -text
app.py CHANGED
@@ -34,10 +34,10 @@ with st.sidebar:
34
 
35
  # --- starter questions (clickable examples) ---------------------------------
36
  STARTER_QUESTIONS = [
37
- "What are AMD's main business risks?",
38
- "What are Abbott's business segments?",
39
- "What is Air Products' primary business?",
40
- "What does Matson's logistics business do?",
41
  ]
42
 
43
  # --- chat history -----------------------------------------------------------
 
34
 
35
  # --- starter questions (clickable examples) ---------------------------------
36
  STARTER_QUESTIONS = [
37
+ "What are Boeing's business segments?",
38
+ "What are the main risks AMD identifies?",
39
+ "What products and services does Microsoft offer?",
40
+ "Which geographies does PepsiCo operate in?",
41
  ]
42
 
43
  # --- chat history -----------------------------------------------------------
eval/run_eval.py CHANGED
@@ -1,9 +1,11 @@
1
- """Evaluate FinChat on the curated gold set with an LLM-as-judge score.
2
 
3
- For each question, FinChat produces an answer, then a separate LLM "judge"
4
- grades that answer against the reference: CORRECT (1.0), PARTIAL (0.5), or
5
- INCORRECT (0.0). Results and the overall accuracy are written to
6
- eval/results.md.
 
 
7
 
8
  Run from the project root:
9
  python -m eval.run_eval
@@ -18,21 +20,54 @@ from pathlib import Path
18
  # Make `src` and `eval` importable no matter how this is launched.
19
  sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
20
 
 
21
  from langchain_core.prompts import ChatPromptTemplate
 
22
 
 
23
  from src.rag import answer, get_llm
24
- from eval.gold_set import GOLD_SET
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
25
 
26
  JUDGE_PROMPT = ChatPromptTemplate.from_messages(
27
  [
28
  (
29
  "system",
30
  "You grade a financial question-answering system against a reference "
31
- "answer.\n"
32
- "Grade CORRECT if the system answer conveys the key facts of the "
33
- "reference (minor omissions or extra detail are fine), PARTIAL if it "
34
- "captures some but misses important parts, and INCORRECT if it is "
35
- "wrong, empty, or unsupported.\n"
36
  "Respond in EXACTLY this format:\n"
37
  "VERDICT: <CORRECT|PARTIAL|INCORRECT>\n"
38
  "REASON: <one short sentence>",
@@ -44,12 +79,11 @@ JUDGE_PROMPT = ChatPromptTemplate.from_messages(
44
  ),
45
  ]
46
  )
47
-
48
  SCORE = {"CORRECT": 1.0, "PARTIAL": 0.5, "INCORRECT": 0.0}
49
 
50
 
51
  def judge(question: str, reference: str, system: str) -> tuple[str, str]:
52
- text = (JUDGE_PROMPT | get_llm()).invoke(
53
  {"question": question, "reference": reference, "system": system}
54
  ).content
55
  v = re.search(r"VERDICT:\s*(CORRECT|PARTIAL|INCORRECT)", text, re.I)
@@ -59,46 +93,85 @@ def judge(question: str, reference: str, system: str) -> tuple[str, str]:
59
  return verdict, reason
60
 
61
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  def main() -> None:
 
 
 
 
 
 
63
  results = []
64
  total = 0.0
65
-
66
- for i, item in enumerate(GOLD_SET, 1):
67
- res = answer(item["question"])
68
- verdict, reason = judge(item["question"], item["reference"], res["answer"])
69
  total += SCORE[verdict]
70
- results.append((item, res, verdict, reason))
71
- print(f"[{verdict:9}] {item['company']:5} {item['question']}")
72
- time.sleep(1.0) # be gentle with the free-tier rate limit
73
 
74
- n = len(GOLD_SET)
75
  accuracy = total / n if n else 0.0
76
- correct = sum(1 for _, _, v, _ in results if v == "CORRECT")
77
- partial = sum(1 for _, _, v, _ in results if v == "PARTIAL")
78
 
79
  lines = [
80
- "# FinChat Evaluation — Curated Gold Set\n",
81
- "FinChat is graded by an LLM-as-judge against reference answers written "
82
- "from the 2017-2020 10-K filings in the corpus.\n",
83
- f"- **Questions:** {n}",
 
 
84
  f"- **CORRECT:** {correct} **PARTIAL:** {partial} "
85
  f"**INCORRECT:** {n - correct - partial}",
86
  f"- **Score:** {total:.1f} / {n}",
87
  f"- **Accuracy (CORRECT=1.0, PARTIAL=0.5):** {accuracy:.0%}\n",
88
- "| # | Company | Verdict | Question |",
89
- "|---|---------|---------|----------|",
90
  ]
91
- for i, (item, _res, verdict, _reason) in enumerate(results, 1):
92
- lines.append(f"| {i} | {item['company']} | {verdict} | {item['question']} |")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
93
 
94
  lines += ["\n---\n", "## Detailed results\n"]
95
- for i, (item, res, verdict, reason) in enumerate(results, 1):
96
  lines += [
97
- f"### {i}. {item['question']}",
98
- f"- **Company:** {item['company']} | "
99
  f"**Routed to:** {res['routed_to']} | **Verdict:** {verdict}",
100
  f"- **Judge:** {reason}",
101
- f"- **Reference:** {item['reference']}",
102
  f"- **FinChat:** {res['answer']}\n",
103
  ]
104
 
 
1
+ """Evaluate FinChat against the FinanceBench benchmark (corpus-aligned subset).
2
 
3
+ Runs FinChat on every FinanceBench 10-K question whose company + fiscal year is
4
+ present in our EDGAR corpus, then grades each answer against FinanceBench's gold
5
+ answer with an LLM-as-judge. Writes eval/results.md.
6
+
7
+ FinanceBench is a deliberately hard, expert-written benchmark, so the goal is an
8
+ honest, measured score on real questions — not a perfect one.
9
 
10
  Run from the project root:
11
  python -m eval.run_eval
 
20
  # Make `src` and `eval` importable no matter how this is launched.
21
  sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
22
 
23
+ from datasets import load_dataset
24
  from langchain_core.prompts import ChatPromptTemplate
25
+ from langchain_groq import ChatGroq
26
 
27
+ from src import config
28
  from src.rag import answer, get_llm
29
+
30
+ # Cap the number of questions so a full run fits Groq's free-tier daily token
31
+ # budget. Sampled with an even stride across companies for representativeness.
32
+ MAX_QUESTIONS = 30
33
+ # Grade with a small, cheap, high-rate-limit model (judging is easy); this
34
+ # keeps the expensive 70B model's token budget for answering.
35
+ JUDGE_MODEL = "llama-3.1-8b-instant"
36
+ _judge_llm = None
37
+
38
+
39
+ def get_judge_llm() -> ChatGroq:
40
+ global _judge_llm
41
+ if _judge_llm is None:
42
+ _judge_llm = ChatGroq(model=JUDGE_MODEL, temperature=0.0, max_retries=5)
43
+ return _judge_llm
44
+
45
+ # Windows consoles default to cp1252 and crash when print() emits Unicode
46
+ # (curly quotes etc. from FinanceBench questions). Force UTF-8 output.
47
+ try:
48
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
49
+ except Exception:
50
+ pass
51
+
52
+ # FinanceBench company name -> ticker (aligned with config.TARGET_FILINGS).
53
+ FB_COMPANY_TO_TICKER = {
54
+ "AMD": "AMD", "American Express": "AXP", "Boeing": "BA", "PepsiCo": "PEP",
55
+ "Amcor": "AMCR", "3M": "MMM", "Johnson & Johnson": "JNJ", "CVS Health": "CVS",
56
+ "Pfizer": "PFE", "AES Corporation": "AES", "Verizon": "VZ", "Best Buy": "BBY",
57
+ "Adobe": "ADBE", "Ulta Beauty": "ULTA", "Coca-Cola": "KO", "Microsoft": "MSFT",
58
+ "Nike": "NKE", "Corning": "GLW",
59
+ }
60
 
61
  JUDGE_PROMPT = ChatPromptTemplate.from_messages(
62
  [
63
  (
64
  "system",
65
  "You grade a financial question-answering system against a reference "
66
+ "answer from the FinanceBench benchmark.\n"
67
+ "Grade CORRECT if the system answer agrees with the reference on the "
68
+ "key fact(s) or number(s) (minor wording or rounding is fine), "
69
+ "PARTIAL if it is partially right or incomplete, and INCORRECT if it "
70
+ "is wrong, empty, or says it cannot find the answer.\n"
71
  "Respond in EXACTLY this format:\n"
72
  "VERDICT: <CORRECT|PARTIAL|INCORRECT>\n"
73
  "REASON: <one short sentence>",
 
79
  ),
80
  ]
81
  )
 
82
  SCORE = {"CORRECT": 1.0, "PARTIAL": 0.5, "INCORRECT": 0.0}
83
 
84
 
85
  def judge(question: str, reference: str, system: str) -> tuple[str, str]:
86
+ text = (JUDGE_PROMPT | get_judge_llm()).invoke(
87
  {"question": question, "reference": reference, "system": system}
88
  ).content
89
  v = re.search(r"VERDICT:\s*(CORRECT|PARTIAL|INCORRECT)", text, re.I)
 
93
  return verdict, reason
94
 
95
 
96
+ def select_questions() -> list[dict]:
97
+ """FinanceBench 10-K questions whose (company, fiscal year) is in our corpus."""
98
+ ingested = {(t, str(y)) for t, _name, y in config.TARGET_FILINGS}
99
+ fb = load_dataset("PatronusAI/financebench", split="train")
100
+ picked = []
101
+ for ex in fb:
102
+ if "10K" not in str(ex.get("doc_name", "")):
103
+ continue
104
+ ticker = FB_COMPANY_TO_TICKER.get(str(ex.get("company")))
105
+ if ticker and (ticker, str(ex.get("doc_period"))) in ingested:
106
+ picked.append(ex)
107
+ return picked
108
+
109
+
110
  def main() -> None:
111
+ rows = select_questions()
112
+ if len(rows) > MAX_QUESTIONS: # even-stride sample
113
+ stride = len(rows) // MAX_QUESTIONS
114
+ rows = rows[::stride][:MAX_QUESTIONS]
115
+ print(f"Evaluating {len(rows)} corpus-aligned FinanceBench 10-K questions.\n")
116
+
117
  results = []
118
  total = 0.0
119
+ for ex in rows:
120
+ res = answer(ex["question"])
121
+ verdict, reason = judge(ex["question"], ex.get("answer", ""), res["answer"])
 
122
  total += SCORE[verdict]
123
+ results.append((ex, res, verdict, reason))
124
+ print(f"[{verdict:9}] {ex.get('company')}: {ex['question'][:70]}")
125
+ time.sleep(1.0) # ease off the free-tier rate limit
126
 
127
+ n = len(rows)
128
  accuracy = total / n if n else 0.0
129
+ correct = sum(1 for _e, _r, v, _j in results if v == "CORRECT")
130
+ partial = sum(1 for _e, _r, v, _j in results if v == "PARTIAL")
131
 
132
  lines = [
133
+ "# FinChat Evaluation — FinanceBench (corpus-aligned subset)\n",
134
+ "FinChat is graded by an LLM-as-judge against gold answers from the "
135
+ "[FinanceBench](https://huggingface.co/datasets/PatronusAI/financebench) "
136
+ "benchmark, on every 10-K question whose company + fiscal year is in the "
137
+ "corpus. FinanceBench is expert-written and intentionally hard.\n",
138
+ f"- **Questions evaluated:** {n}",
139
  f"- **CORRECT:** {correct} **PARTIAL:** {partial} "
140
  f"**INCORRECT:** {n - correct - partial}",
141
  f"- **Score:** {total:.1f} / {n}",
142
  f"- **Accuracy (CORRECT=1.0, PARTIAL=0.5):** {accuracy:.0%}\n",
 
 
143
  ]
144
+
145
+ # Accuracy by FinanceBench question type -> shows the qualitative-vs-numeric
146
+ # split (metrics-generated questions require computation over tables).
147
+ by_type: dict[str, list[float]] = {}
148
+ for ex, _res, verdict, _reason in results:
149
+ agg = by_type.setdefault(str(ex.get("question_type", "unknown")), [0.0, 0])
150
+ agg[0] += SCORE[verdict]
151
+ agg[1] += 1
152
+ print("\nAccuracy by question type:")
153
+ lines += ["**Accuracy by FinanceBench question type:**\n",
154
+ "| Question type | Accuracy | N |", "|---|---|---|"]
155
+ for qt, (s, c) in sorted(by_type.items()):
156
+ print(f" {qt:24} {s / c:.0%} ({c})")
157
+ lines.append(f"| {qt} | {s / c:.0%} | {c} |")
158
+
159
+ lines += ["\n| # | Company | FY | Verdict | Question |",
160
+ "|---|---------|----|---------|----------|"]
161
+ for i, (ex, _res, verdict, _reason) in enumerate(results, 1):
162
+ q = ex["question"].replace("|", "\\|")
163
+ lines.append(
164
+ f"| {i} | {ex.get('company')} | {ex.get('doc_period')} | {verdict} | {q} |"
165
+ )
166
 
167
  lines += ["\n---\n", "## Detailed results\n"]
168
+ for i, (ex, res, verdict, reason) in enumerate(results, 1):
169
  lines += [
170
+ f"### {i}. {ex['question']}",
171
+ f"- **Company / FY:** {ex.get('company')} {ex.get('doc_period')} | "
172
  f"**Routed to:** {res['routed_to']} | **Verdict:** {verdict}",
173
  f"- **Judge:** {reason}",
174
+ f"- **FinanceBench gold:** {ex.get('answer', '')}",
175
  f"- **FinChat:** {res['answer']}\n",
176
  ]
177
 
requirements.txt CHANGED
@@ -18,8 +18,11 @@ chromadb==1.5.9
18
  # --- Embedding model runtime (pulls in torch, CPU is fine) ---
19
  sentence-transformers==5.6.0
20
 
21
- # --- Data loading ---
22
- # datasets must stay <3.0: this dataset ships a loader script, and
 
 
 
23
  # datasets>=3.0 removed support for script-based datasets.
24
  datasets==2.21.0
25
  pandas==3.0.3
 
18
  # --- Embedding model runtime (pulls in torch, CPU is fine) ---
19
  sentence-transformers==5.6.0
20
 
21
+ # --- SEC filings source ---
22
+ edgartools==5.40.1 # fetches & extracts 10-K text from SEC EDGAR
23
+
24
+ # --- Data loading (FinanceBench evaluation set) ---
25
+ # datasets must stay <3.0: FinanceBench-adjacent script datasets need it, and
26
  # datasets>=3.0 removed support for script-based datasets.
27
  datasets==2.21.0
28
  pandas==3.0.3
src/config.py CHANGED
@@ -18,30 +18,51 @@ DATA_DIR = PROJECT_ROOT / "data"
18
  # writable, which makes chromadb's storage engine fail to start -> use
19
  # /tmp, which is writable in any container.
20
  # Override either default with the FINCHAT_VECTORSTORE env var.
 
21
  _default_store = (
22
  Path.home() / ".finchat" / "vectorstore"
23
  if os.name == "nt"
24
  else Path("/tmp/finchat/vectorstore")
25
  )
26
- VECTORSTORE_DIR = Path(os.getenv("FINCHAT_VECTORSTORE", str(_default_store)))
27
-
28
- # --- Dataset (Hugging Face) -------------------------------------------------
29
- # Each row is ONE sentence from a 10-K filing. ingest.py reassembles the
30
- # sentences into section text before chunking. "small_full" (~240k sentences)
31
- # keeps the download light enough for a laptop.
32
- HF_DATASET = "JanosAudran/financial-reports-sec"
33
- HF_CONFIG = "small_full"
34
- HF_SPLIT = "train"
35
-
36
- # --- Corpus scope -----------------------------------------------------------
37
- # Which companies to ingest. Leave TARGET_TICKERS = [] to auto-pick the
38
- # TOP_N_COMPANIES with the most content (handy before you know what's in the
39
- # data -- run `python -m src.ingest --list` to see the options).
40
- # Recognizable companies available in the "small_full" config, recent years.
41
- # Swap freely: also available -> CECE, BKTI, ACU, AE, WDDD (all end at 2020).
42
- TARGET_TICKERS: list[str] = ["AMD", "ABT", "APD", "AIR", "MATX"]
43
- TARGET_YEARS: list[int] = [2017, 2018, 2019, 2020]
44
- TOP_N_COMPANIES = 8 # used only when TARGET_TICKERS is empty
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
45
 
46
  # --- Chunking ---------------------------------------------------------------
47
  CHUNK_SIZE = 900 # characters per chunk
 
18
  # writable, which makes chromadb's storage engine fail to start -> use
19
  # /tmp, which is writable in any container.
20
  # Override either default with the FINCHAT_VECTORSTORE env var.
21
+ _repo_store = PROJECT_ROOT / "vectorstore" # prebuilt index committed to repo
22
  _default_store = (
23
  Path.home() / ".finchat" / "vectorstore"
24
  if os.name == "nt"
25
  else Path("/tmp/finchat/vectorstore")
26
  )
27
+ if os.getenv("FINCHAT_VECTORSTORE"):
28
+ VECTORSTORE_DIR = Path(os.environ["FINCHAT_VECTORSTORE"])
29
+ elif (_repo_store / "chroma.sqlite3").exists():
30
+ # A prebuilt index shipped with the repo (e.g. on Hugging Face Spaces) —
31
+ # use it directly so the app never rebuilds on startup.
32
+ VECTORSTORE_DIR = _repo_store
33
+ else:
34
+ VECTORSTORE_DIR = _default_store
35
+
36
+ # --- SEC EDGAR corpus -------------------------------------------------------
37
+ # Recent 10-K filings for recognizable companies, fetched from SEC EDGAR via
38
+ # edgartools. The set overlaps with the FinanceBench benchmark (matching
39
+ # company + fiscal year), so FinChat can be scored against it.
40
+ # SEC requires a contact identity (name/email) on every request.
41
+ EDGAR_IDENTITY = "narendra.daffa08@gmail.com"
42
+
43
+ # (ticker, display name, fiscal year). fiscal_year matches the filing's
44
+ # period_of_report year, which correctly handles offset fiscal years
45
+ # (e.g. Amcor closes in June, Nike in May).
46
+ TARGET_FILINGS = [
47
+ ("AMD", "Advanced Micro Devices", 2022),
48
+ ("AXP", "American Express", 2022),
49
+ ("BA", "Boeing", 2022),
50
+ ("PEP", "PepsiCo", 2022),
51
+ ("AMCR", "Amcor", 2023),
52
+ ("MMM", "3M", 2022),
53
+ ("JNJ", "Johnson & Johnson", 2022),
54
+ ("CVS", "CVS Health", 2022),
55
+ ("PFE", "Pfizer", 2021),
56
+ ("AES", "AES Corporation", 2022),
57
+ ("VZ", "Verizon", 2022),
58
+ ("BBY", "Best Buy", 2023),
59
+ ("ADBE", "Adobe", 2022),
60
+ ("ULTA", "Ulta Beauty", 2023),
61
+ ("KO", "Coca-Cola", 2022),
62
+ ("MSFT", "Microsoft", 2023),
63
+ ("NKE", "Nike", 2023),
64
+ ("GLW", "Corning", 2022),
65
+ ]
66
 
67
  # --- Chunking ---------------------------------------------------------------
68
  CHUNK_SIZE = 900 # characters per chunk
src/ingest.py CHANGED
@@ -1,17 +1,14 @@
1
- """Build the FinChat vector store from SEC 10-K filings.
2
 
3
  Pipeline:
4
- load sentences -> group into section text -> split into chunks
5
- -> embed locally -> store in Chroma (persisted to ./vectorstore)
6
 
7
- Run it once (from the project root):
8
- python -m src.ingest # build the vector store
9
- python -m src.ingest --list # just list available companies, then exit
10
  """
11
  from __future__ import annotations
12
 
13
- import argparse
14
- import re
15
  import shutil
16
  import sys
17
  import time
@@ -20,8 +17,7 @@ from pathlib import Path
20
  # Allow running as either `python -m src.ingest` or `python src/ingest.py`.
21
  sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
22
 
23
- import pandas as pd
24
- from datasets import load_dataset
25
  from langchain_core.documents import Document
26
  from langchain_text_splitters import RecursiveCharacterTextSplitter
27
  from langchain_huggingface import HuggingFaceEmbeddings
@@ -29,105 +25,60 @@ from langchain_chroma import Chroma
29
 
30
  from src import config
31
 
 
 
 
 
 
 
32
 
33
- def pretty_section(label: str) -> str:
34
- """Turn a raw section label into a readable 'Section N' for display."""
35
- nums = re.findall(r"\d+", str(label))
36
- if nums:
37
- return f"Section {nums[-1]}"
38
- return str(label).replace("_", " ").strip().title() or "Filing"
39
 
 
 
40
 
41
- def load_dataframe() -> pd.DataFrame:
42
- """Download the dataset split and return it as a tidy DataFrame."""
43
- print(f"Loading {config.HF_DATASET} [{config.HF_CONFIG}] ... (first run downloads it)")
44
- ds = load_dataset(
45
- config.HF_DATASET, config.HF_CONFIG, split=config.HF_SPLIT,
46
- trust_remote_code=True,
47
- )
48
- df = ds.to_pandas()
49
 
50
- # `tickers` is a list per row -> take the first as the primary ticker.
51
- df["ticker"] = df["tickers"].apply(
52
- lambda t: t[0] if hasattr(t, "__len__") and len(t) else None
53
- )
54
- # reportDate looks like "2020-09-26"; the year is the first 4 chars.
55
- df["year"] = df["reportDate"].astype(str).str.slice(0, 4)
56
- return df
57
-
58
-
59
- def list_companies(df: pd.DataFrame, top: int = 30) -> None:
60
- counts = (
61
- df.dropna(subset=["ticker"])
62
- .groupby(["ticker", "name"])
63
- .size()
64
- .sort_values(ascending=False)
65
- .head(top)
66
- )
67
- print("\nTop companies available (ticker | name | #sentences):")
68
- for (ticker, name), n in counts.items():
69
- print(f" {ticker:<8} {str(name):<42} {n}")
70
- print("\nCopy the tickers you want into TARGET_TICKERS in src/config.py.")
71
-
72
-
73
- def select_rows(df: pd.DataFrame) -> pd.DataFrame:
74
- df = df.dropna(subset=["ticker"]).copy()
75
-
76
- if config.TARGET_YEARS:
77
- years = {str(y) for y in config.TARGET_YEARS}
78
- df = df[df["year"].isin(years)]
79
-
80
- if config.TARGET_TICKERS:
81
- wanted = {t.upper() for t in config.TARGET_TICKERS}
82
- df = df[df["ticker"].str.upper().isin(wanted)]
83
- else:
84
- # No explicit list -> keep the TOP_N companies by content volume.
85
- top = (
86
- df.groupby("ticker").size()
87
- .sort_values(ascending=False)
88
- .head(config.TOP_N_COMPANIES)
89
- .index
90
- )
91
- df = df[df["ticker"].isin(top)]
92
-
93
- return df
94
-
95
-
96
- def build_documents(df: pd.DataFrame) -> list[Document]:
97
- """Reassemble sentences into section text, then split into chunks."""
98
- # One "raw document" per (filing, section). docID identifies one filing.
99
- grouped = df.sort_values("sentenceCount").groupby(["docID", "section"])
100
-
101
- raw_docs: list[Document] = []
102
- for (doc_id, section), rows in grouped:
103
- text = " ".join(str(s) for s in rows["sentence"].tolist()).strip()
104
- if len(text) < 50: # skip near-empty sections
105
- continue
106
- head = rows.iloc[0]
107
- company = str(head["name"])
108
- year = str(head["year"])
109
- raw_docs.append(
110
- Document(
111
- page_content=text,
112
- metadata={
113
- "ticker": str(head["ticker"]),
114
- "company": company,
115
- "year": year,
116
- "section": str(section),
117
- "cik": str(head["cik"]),
118
- "docID": str(doc_id),
119
- "source": f"{company} 10-K ({year}) - {pretty_section(section)}",
120
- },
121
- )
122
- )
123
 
 
 
 
124
  splitter = RecursiveCharacterTextSplitter(
125
  chunk_size=config.CHUNK_SIZE,
126
  chunk_overlap=config.CHUNK_OVERLAP,
127
  )
128
- chunks = splitter.split_documents(raw_docs)
129
- print(f"Reassembled {len(raw_docs)} sections -> {len(chunks)} chunks.")
130
- return chunks
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
131
 
132
 
133
  def _safe_rmtree(path: Path, retries: int = 3) -> None:
@@ -138,7 +89,7 @@ def _safe_rmtree(path: Path, retries: int = 3) -> None:
138
  return
139
  except (PermissionError, OSError):
140
  time.sleep(1.0)
141
- shutil.rmtree(path) # last try -- let the error surface if it still fails
142
 
143
 
144
  def build_vectorstore(chunks: list[Document]) -> None:
@@ -150,7 +101,7 @@ def build_vectorstore(chunks: list[Document]) -> None:
150
  print(f"Loading embedding model {config.EMBEDDING_MODEL} ...")
151
  embeddings = HuggingFaceEmbeddings(model_name=config.EMBEDDING_MODEL)
152
 
153
- print("Embedding & storing chunks (this can take a few minutes) ...")
154
  Chroma.from_documents(
155
  documents=chunks,
156
  embedding=embeddings,
@@ -161,34 +112,18 @@ def build_vectorstore(chunks: list[Document]) -> None:
161
 
162
 
163
  def build_index() -> None:
164
- """Run the full ingestion pipeline: load -> select -> chunk -> embed -> store.
165
 
166
- Importable so the app can bootstrap the vector store on first run
167
- (e.g. on a fresh Hugging Face Space).
168
  """
169
- df = load_dataframe()
170
- selected = select_rows(df)
171
- if selected.empty:
172
- raise SystemExit(
173
- "\nNo rows matched TARGET_TICKERS / TARGET_YEARS in src/config.py.\n"
174
- "Run `python -m src.ingest --list` to see what's available."
175
- )
176
- companies = sorted(selected["ticker"].unique())
177
- print(f"Ingesting {len(companies)} companies: {', '.join(companies)}")
178
- chunks = build_documents(selected)
179
- build_vectorstore(chunks)
180
 
181
 
182
  def main() -> None:
183
- parser = argparse.ArgumentParser(description="Build the FinChat vector store.")
184
- parser.add_argument("--list", action="store_true",
185
- help="List available companies and exit.")
186
- args = parser.parse_args()
187
-
188
- if args.list:
189
- list_companies(load_dataframe())
190
- return
191
-
192
  build_index()
193
 
194
 
 
1
+ """Build the FinChat vector store from recent SEC 10-K filings (EDGAR).
2
 
3
  Pipeline:
4
+ fetch each target 10-K (edgartools) -> split into chunks
5
+ -> embed locally -> store in Chroma (persisted to config.VECTORSTORE_DIR)
6
 
7
+ Run once, from the project root:
8
+ python -m src.ingest
 
9
  """
10
  from __future__ import annotations
11
 
 
 
12
  import shutil
13
  import sys
14
  import time
 
17
  # Allow running as either `python -m src.ingest` or `python src/ingest.py`.
18
  sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
19
 
20
+ from edgar import Company, set_identity
 
21
  from langchain_core.documents import Document
22
  from langchain_text_splitters import RecursiveCharacterTextSplitter
23
  from langchain_huggingface import HuggingFaceEmbeddings
 
25
 
26
  from src import config
27
 
28
+ # Windows consoles default to cp1252 and crash when print() emits Unicode
29
+ # (arrows, em-dashes, curly quotes from filings). Force UTF-8 output.
30
+ try:
31
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
32
+ except Exception:
33
+ pass
34
 
 
 
 
 
 
 
35
 
36
+ def fetch_filing(ticker: str, fiscal_year: int) -> tuple[str, str] | None:
37
+ """Return (text, accession_no) for the 10-K whose fiscal period matches.
38
 
39
+ Matches on the filing's period_of_report year, so an offset fiscal year
40
+ (e.g. Amcor's June close) still resolves to the right filing.
41
+ """
42
+ for f in Company(ticker).get_filings(form="10-K"):
43
+ period = getattr(f, "period_of_report", None)
44
+ if period and str(period)[:4] == str(fiscal_year):
45
+ return f.text(), f.accession_no
46
+ return None
47
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
48
 
49
+ def fetch_documents() -> list[Document]:
50
+ """Fetch every target filing and split it into chunked LangChain Documents."""
51
+ set_identity(config.EDGAR_IDENTITY)
52
  splitter = RecursiveCharacterTextSplitter(
53
  chunk_size=config.CHUNK_SIZE,
54
  chunk_overlap=config.CHUNK_OVERLAP,
55
  )
56
+
57
+ docs: list[Document] = []
58
+ for ticker, name, year in config.TARGET_FILINGS:
59
+ print(f"Fetching {ticker} FY{year} 10-K ...", end=" ", flush=True)
60
+ result = fetch_filing(ticker, year)
61
+ if result is None:
62
+ print("NOT FOUND -- skipping")
63
+ continue
64
+ text, accession = result
65
+ source = f"{name} 10-K (FY{year})"
66
+ for chunk in splitter.split_text(text):
67
+ docs.append(
68
+ Document(
69
+ page_content=chunk,
70
+ metadata={
71
+ "ticker": ticker,
72
+ "company": name,
73
+ "year": str(year),
74
+ "accession": accession,
75
+ "source": source,
76
+ },
77
+ )
78
+ )
79
+ print(f"{len(text):,} chars -> {len(docs)} chunks so far")
80
+ time.sleep(0.5) # be polite to SEC's servers
81
+ return docs
82
 
83
 
84
  def _safe_rmtree(path: Path, retries: int = 3) -> None:
 
89
  return
90
  except (PermissionError, OSError):
91
  time.sleep(1.0)
92
+ shutil.rmtree(path)
93
 
94
 
95
  def build_vectorstore(chunks: list[Document]) -> None:
 
101
  print(f"Loading embedding model {config.EMBEDDING_MODEL} ...")
102
  embeddings = HuggingFaceEmbeddings(model_name=config.EMBEDDING_MODEL)
103
 
104
+ print(f"Embedding & storing {len(chunks)} chunks (a few minutes) ...")
105
  Chroma.from_documents(
106
  documents=chunks,
107
  embedding=embeddings,
 
112
 
113
 
114
  def build_index() -> None:
115
+ """Full pipeline: fetch filings -> chunk -> embed -> store.
116
 
117
+ Importable so the app can bootstrap the store on first run.
 
118
  """
119
+ docs = fetch_documents()
120
+ if not docs:
121
+ raise SystemExit("No filings were fetched — check tickers/years in config.py.")
122
+ print(f"Total: {len(docs)} chunks from {len(config.TARGET_FILINGS)} target filings.")
123
+ build_vectorstore(docs)
 
 
 
 
 
 
124
 
125
 
126
  def main() -> None:
 
 
 
 
 
 
 
 
 
127
  build_index()
128
 
129
 
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/data_level0.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6b529e7f484c17ad5a211753037d3fd28fd20e1ccbd9f9358289b6a69c4be31
3
+ size 27018796
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/header.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bcdaa3a4152103e1edc94b56aad83a3557a6828134c42d68f9260717e1f9f243
3
+ size 100
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/index_metadata.pickle ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54c22b54ab9f504cadfc301b506ca9f2b0b7769736bc624af5fa2402d09716cc
3
+ size 1483324
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/length.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:034a92141994e5554a9ca608b1f904f4c4a35b65799fffad10c0086c0505943e
3
+ size 64484
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/link_lists.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ddeffe7ec74e3fc7fbdf981a4b7e3d733cdcaabe2aa54b6446ccd870654afe9
3
+ size 139012
vectorstore/chroma.sqlite3 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ab762371a44ad19ed423c7b6c4d3b94432832e17dca56a2cab388d02882ebe1
3
+ size 99049472