Expand corpus to 18 companies (recent SEC 10-Ks); ship prebuilt index via git-lfs
Browse files- .gitattributes +1 -0
- app.py +4 -4
- eval/run_eval.py +108 -35
- requirements.txt +5 -2
- src/config.py +40 -19
- src/ingest.py +60 -125
- vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/data_level0.bin +3 -0
- vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/header.bin +3 -0
- vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/index_metadata.pickle +3 -0
- vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/length.bin +3 -0
- vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/link_lists.bin +3 -0
- vectorstore/chroma.sqlite3 +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
vectorstore/** filter=lfs diff=lfs merge=lfs -text
|
app.py
CHANGED
|
@@ -34,10 +34,10 @@ with st.sidebar:
|
|
| 34 |
|
| 35 |
# --- starter questions (clickable examples) ---------------------------------
|
| 36 |
STARTER_QUESTIONS = [
|
| 37 |
-
"What are
|
| 38 |
-
"What are
|
| 39 |
-
"What
|
| 40 |
-
"
|
| 41 |
]
|
| 42 |
|
| 43 |
# --- chat history -----------------------------------------------------------
|
|
|
|
| 34 |
|
| 35 |
# --- starter questions (clickable examples) ---------------------------------
|
| 36 |
STARTER_QUESTIONS = [
|
| 37 |
+
"What are Boeing's business segments?",
|
| 38 |
+
"What are the main risks AMD identifies?",
|
| 39 |
+
"What products and services does Microsoft offer?",
|
| 40 |
+
"Which geographies does PepsiCo operate in?",
|
| 41 |
]
|
| 42 |
|
| 43 |
# --- chat history -----------------------------------------------------------
|
eval/run_eval.py
CHANGED
|
@@ -1,9 +1,11 @@
|
|
| 1 |
-
"""Evaluate FinChat
|
| 2 |
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
|
|
|
|
|
|
| 7 |
|
| 8 |
Run from the project root:
|
| 9 |
python -m eval.run_eval
|
|
@@ -18,21 +20,54 @@ from pathlib import Path
|
|
| 18 |
# Make `src` and `eval` importable no matter how this is launched.
|
| 19 |
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
| 20 |
|
|
|
|
| 21 |
from langchain_core.prompts import ChatPromptTemplate
|
|
|
|
| 22 |
|
|
|
|
| 23 |
from src.rag import answer, get_llm
|
| 24 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 25 |
|
| 26 |
JUDGE_PROMPT = ChatPromptTemplate.from_messages(
|
| 27 |
[
|
| 28 |
(
|
| 29 |
"system",
|
| 30 |
"You grade a financial question-answering system against a reference "
|
| 31 |
-
"answer.\n"
|
| 32 |
-
"Grade CORRECT if the system answer
|
| 33 |
-
"
|
| 34 |
-
"
|
| 35 |
-
"wrong, empty, or
|
| 36 |
"Respond in EXACTLY this format:\n"
|
| 37 |
"VERDICT: <CORRECT|PARTIAL|INCORRECT>\n"
|
| 38 |
"REASON: <one short sentence>",
|
|
@@ -44,12 +79,11 @@ JUDGE_PROMPT = ChatPromptTemplate.from_messages(
|
|
| 44 |
),
|
| 45 |
]
|
| 46 |
)
|
| 47 |
-
|
| 48 |
SCORE = {"CORRECT": 1.0, "PARTIAL": 0.5, "INCORRECT": 0.0}
|
| 49 |
|
| 50 |
|
| 51 |
def judge(question: str, reference: str, system: str) -> tuple[str, str]:
|
| 52 |
-
text = (JUDGE_PROMPT |
|
| 53 |
{"question": question, "reference": reference, "system": system}
|
| 54 |
).content
|
| 55 |
v = re.search(r"VERDICT:\s*(CORRECT|PARTIAL|INCORRECT)", text, re.I)
|
|
@@ -59,46 +93,85 @@ def judge(question: str, reference: str, system: str) -> tuple[str, str]:
|
|
| 59 |
return verdict, reason
|
| 60 |
|
| 61 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
def main() -> None:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 63 |
results = []
|
| 64 |
total = 0.0
|
| 65 |
-
|
| 66 |
-
|
| 67 |
-
|
| 68 |
-
verdict, reason = judge(item["question"], item["reference"], res["answer"])
|
| 69 |
total += SCORE[verdict]
|
| 70 |
-
results.append((
|
| 71 |
-
print(f"[{verdict:9}] {
|
| 72 |
-
time.sleep(1.0) #
|
| 73 |
|
| 74 |
-
n = len(
|
| 75 |
accuracy = total / n if n else 0.0
|
| 76 |
-
correct = sum(1 for
|
| 77 |
-
partial = sum(1 for
|
| 78 |
|
| 79 |
lines = [
|
| 80 |
-
"# FinChat Evaluation —
|
| 81 |
-
"FinChat is graded by an LLM-as-judge against
|
| 82 |
-
"
|
| 83 |
-
|
|
|
|
|
|
|
| 84 |
f"- **CORRECT:** {correct} **PARTIAL:** {partial} "
|
| 85 |
f"**INCORRECT:** {n - correct - partial}",
|
| 86 |
f"- **Score:** {total:.1f} / {n}",
|
| 87 |
f"- **Accuracy (CORRECT=1.0, PARTIAL=0.5):** {accuracy:.0%}\n",
|
| 88 |
-
"| # | Company | Verdict | Question |",
|
| 89 |
-
"|---|---------|---------|----------|",
|
| 90 |
]
|
| 91 |
-
|
| 92 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 93 |
|
| 94 |
lines += ["\n---\n", "## Detailed results\n"]
|
| 95 |
-
for i, (
|
| 96 |
lines += [
|
| 97 |
-
f"### {i}. {
|
| 98 |
-
f"- **Company:** {
|
| 99 |
f"**Routed to:** {res['routed_to']} | **Verdict:** {verdict}",
|
| 100 |
f"- **Judge:** {reason}",
|
| 101 |
-
f"- **
|
| 102 |
f"- **FinChat:** {res['answer']}\n",
|
| 103 |
]
|
| 104 |
|
|
|
|
| 1 |
+
"""Evaluate FinChat against the FinanceBench benchmark (corpus-aligned subset).
|
| 2 |
|
| 3 |
+
Runs FinChat on every FinanceBench 10-K question whose company + fiscal year is
|
| 4 |
+
present in our EDGAR corpus, then grades each answer against FinanceBench's gold
|
| 5 |
+
answer with an LLM-as-judge. Writes eval/results.md.
|
| 6 |
+
|
| 7 |
+
FinanceBench is a deliberately hard, expert-written benchmark, so the goal is an
|
| 8 |
+
honest, measured score on real questions — not a perfect one.
|
| 9 |
|
| 10 |
Run from the project root:
|
| 11 |
python -m eval.run_eval
|
|
|
|
| 20 |
# Make `src` and `eval` importable no matter how this is launched.
|
| 21 |
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
| 22 |
|
| 23 |
+
from datasets import load_dataset
|
| 24 |
from langchain_core.prompts import ChatPromptTemplate
|
| 25 |
+
from langchain_groq import ChatGroq
|
| 26 |
|
| 27 |
+
from src import config
|
| 28 |
from src.rag import answer, get_llm
|
| 29 |
+
|
| 30 |
+
# Cap the number of questions so a full run fits Groq's free-tier daily token
|
| 31 |
+
# budget. Sampled with an even stride across companies for representativeness.
|
| 32 |
+
MAX_QUESTIONS = 30
|
| 33 |
+
# Grade with a small, cheap, high-rate-limit model (judging is easy); this
|
| 34 |
+
# keeps the expensive 70B model's token budget for answering.
|
| 35 |
+
JUDGE_MODEL = "llama-3.1-8b-instant"
|
| 36 |
+
_judge_llm = None
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def get_judge_llm() -> ChatGroq:
|
| 40 |
+
global _judge_llm
|
| 41 |
+
if _judge_llm is None:
|
| 42 |
+
_judge_llm = ChatGroq(model=JUDGE_MODEL, temperature=0.0, max_retries=5)
|
| 43 |
+
return _judge_llm
|
| 44 |
+
|
| 45 |
+
# Windows consoles default to cp1252 and crash when print() emits Unicode
|
| 46 |
+
# (curly quotes etc. from FinanceBench questions). Force UTF-8 output.
|
| 47 |
+
try:
|
| 48 |
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
| 49 |
+
except Exception:
|
| 50 |
+
pass
|
| 51 |
+
|
| 52 |
+
# FinanceBench company name -> ticker (aligned with config.TARGET_FILINGS).
|
| 53 |
+
FB_COMPANY_TO_TICKER = {
|
| 54 |
+
"AMD": "AMD", "American Express": "AXP", "Boeing": "BA", "PepsiCo": "PEP",
|
| 55 |
+
"Amcor": "AMCR", "3M": "MMM", "Johnson & Johnson": "JNJ", "CVS Health": "CVS",
|
| 56 |
+
"Pfizer": "PFE", "AES Corporation": "AES", "Verizon": "VZ", "Best Buy": "BBY",
|
| 57 |
+
"Adobe": "ADBE", "Ulta Beauty": "ULTA", "Coca-Cola": "KO", "Microsoft": "MSFT",
|
| 58 |
+
"Nike": "NKE", "Corning": "GLW",
|
| 59 |
+
}
|
| 60 |
|
| 61 |
JUDGE_PROMPT = ChatPromptTemplate.from_messages(
|
| 62 |
[
|
| 63 |
(
|
| 64 |
"system",
|
| 65 |
"You grade a financial question-answering system against a reference "
|
| 66 |
+
"answer from the FinanceBench benchmark.\n"
|
| 67 |
+
"Grade CORRECT if the system answer agrees with the reference on the "
|
| 68 |
+
"key fact(s) or number(s) (minor wording or rounding is fine), "
|
| 69 |
+
"PARTIAL if it is partially right or incomplete, and INCORRECT if it "
|
| 70 |
+
"is wrong, empty, or says it cannot find the answer.\n"
|
| 71 |
"Respond in EXACTLY this format:\n"
|
| 72 |
"VERDICT: <CORRECT|PARTIAL|INCORRECT>\n"
|
| 73 |
"REASON: <one short sentence>",
|
|
|
|
| 79 |
),
|
| 80 |
]
|
| 81 |
)
|
|
|
|
| 82 |
SCORE = {"CORRECT": 1.0, "PARTIAL": 0.5, "INCORRECT": 0.0}
|
| 83 |
|
| 84 |
|
| 85 |
def judge(question: str, reference: str, system: str) -> tuple[str, str]:
|
| 86 |
+
text = (JUDGE_PROMPT | get_judge_llm()).invoke(
|
| 87 |
{"question": question, "reference": reference, "system": system}
|
| 88 |
).content
|
| 89 |
v = re.search(r"VERDICT:\s*(CORRECT|PARTIAL|INCORRECT)", text, re.I)
|
|
|
|
| 93 |
return verdict, reason
|
| 94 |
|
| 95 |
|
| 96 |
+
def select_questions() -> list[dict]:
|
| 97 |
+
"""FinanceBench 10-K questions whose (company, fiscal year) is in our corpus."""
|
| 98 |
+
ingested = {(t, str(y)) for t, _name, y in config.TARGET_FILINGS}
|
| 99 |
+
fb = load_dataset("PatronusAI/financebench", split="train")
|
| 100 |
+
picked = []
|
| 101 |
+
for ex in fb:
|
| 102 |
+
if "10K" not in str(ex.get("doc_name", "")):
|
| 103 |
+
continue
|
| 104 |
+
ticker = FB_COMPANY_TO_TICKER.get(str(ex.get("company")))
|
| 105 |
+
if ticker and (ticker, str(ex.get("doc_period"))) in ingested:
|
| 106 |
+
picked.append(ex)
|
| 107 |
+
return picked
|
| 108 |
+
|
| 109 |
+
|
| 110 |
def main() -> None:
|
| 111 |
+
rows = select_questions()
|
| 112 |
+
if len(rows) > MAX_QUESTIONS: # even-stride sample
|
| 113 |
+
stride = len(rows) // MAX_QUESTIONS
|
| 114 |
+
rows = rows[::stride][:MAX_QUESTIONS]
|
| 115 |
+
print(f"Evaluating {len(rows)} corpus-aligned FinanceBench 10-K questions.\n")
|
| 116 |
+
|
| 117 |
results = []
|
| 118 |
total = 0.0
|
| 119 |
+
for ex in rows:
|
| 120 |
+
res = answer(ex["question"])
|
| 121 |
+
verdict, reason = judge(ex["question"], ex.get("answer", ""), res["answer"])
|
|
|
|
| 122 |
total += SCORE[verdict]
|
| 123 |
+
results.append((ex, res, verdict, reason))
|
| 124 |
+
print(f"[{verdict:9}] {ex.get('company')}: {ex['question'][:70]}")
|
| 125 |
+
time.sleep(1.0) # ease off the free-tier rate limit
|
| 126 |
|
| 127 |
+
n = len(rows)
|
| 128 |
accuracy = total / n if n else 0.0
|
| 129 |
+
correct = sum(1 for _e, _r, v, _j in results if v == "CORRECT")
|
| 130 |
+
partial = sum(1 for _e, _r, v, _j in results if v == "PARTIAL")
|
| 131 |
|
| 132 |
lines = [
|
| 133 |
+
"# FinChat Evaluation — FinanceBench (corpus-aligned subset)\n",
|
| 134 |
+
"FinChat is graded by an LLM-as-judge against gold answers from the "
|
| 135 |
+
"[FinanceBench](https://huggingface.co/datasets/PatronusAI/financebench) "
|
| 136 |
+
"benchmark, on every 10-K question whose company + fiscal year is in the "
|
| 137 |
+
"corpus. FinanceBench is expert-written and intentionally hard.\n",
|
| 138 |
+
f"- **Questions evaluated:** {n}",
|
| 139 |
f"- **CORRECT:** {correct} **PARTIAL:** {partial} "
|
| 140 |
f"**INCORRECT:** {n - correct - partial}",
|
| 141 |
f"- **Score:** {total:.1f} / {n}",
|
| 142 |
f"- **Accuracy (CORRECT=1.0, PARTIAL=0.5):** {accuracy:.0%}\n",
|
|
|
|
|
|
|
| 143 |
]
|
| 144 |
+
|
| 145 |
+
# Accuracy by FinanceBench question type -> shows the qualitative-vs-numeric
|
| 146 |
+
# split (metrics-generated questions require computation over tables).
|
| 147 |
+
by_type: dict[str, list[float]] = {}
|
| 148 |
+
for ex, _res, verdict, _reason in results:
|
| 149 |
+
agg = by_type.setdefault(str(ex.get("question_type", "unknown")), [0.0, 0])
|
| 150 |
+
agg[0] += SCORE[verdict]
|
| 151 |
+
agg[1] += 1
|
| 152 |
+
print("\nAccuracy by question type:")
|
| 153 |
+
lines += ["**Accuracy by FinanceBench question type:**\n",
|
| 154 |
+
"| Question type | Accuracy | N |", "|---|---|---|"]
|
| 155 |
+
for qt, (s, c) in sorted(by_type.items()):
|
| 156 |
+
print(f" {qt:24} {s / c:.0%} ({c})")
|
| 157 |
+
lines.append(f"| {qt} | {s / c:.0%} | {c} |")
|
| 158 |
+
|
| 159 |
+
lines += ["\n| # | Company | FY | Verdict | Question |",
|
| 160 |
+
"|---|---------|----|---------|----------|"]
|
| 161 |
+
for i, (ex, _res, verdict, _reason) in enumerate(results, 1):
|
| 162 |
+
q = ex["question"].replace("|", "\\|")
|
| 163 |
+
lines.append(
|
| 164 |
+
f"| {i} | {ex.get('company')} | {ex.get('doc_period')} | {verdict} | {q} |"
|
| 165 |
+
)
|
| 166 |
|
| 167 |
lines += ["\n---\n", "## Detailed results\n"]
|
| 168 |
+
for i, (ex, res, verdict, reason) in enumerate(results, 1):
|
| 169 |
lines += [
|
| 170 |
+
f"### {i}. {ex['question']}",
|
| 171 |
+
f"- **Company / FY:** {ex.get('company')} {ex.get('doc_period')} | "
|
| 172 |
f"**Routed to:** {res['routed_to']} | **Verdict:** {verdict}",
|
| 173 |
f"- **Judge:** {reason}",
|
| 174 |
+
f"- **FinanceBench gold:** {ex.get('answer', '')}",
|
| 175 |
f"- **FinChat:** {res['answer']}\n",
|
| 176 |
]
|
| 177 |
|
requirements.txt
CHANGED
|
@@ -18,8 +18,11 @@ chromadb==1.5.9
|
|
| 18 |
# --- Embedding model runtime (pulls in torch, CPU is fine) ---
|
| 19 |
sentence-transformers==5.6.0
|
| 20 |
|
| 21 |
-
# ---
|
| 22 |
-
#
|
|
|
|
|
|
|
|
|
|
| 23 |
# datasets>=3.0 removed support for script-based datasets.
|
| 24 |
datasets==2.21.0
|
| 25 |
pandas==3.0.3
|
|
|
|
| 18 |
# --- Embedding model runtime (pulls in torch, CPU is fine) ---
|
| 19 |
sentence-transformers==5.6.0
|
| 20 |
|
| 21 |
+
# --- SEC filings source ---
|
| 22 |
+
edgartools==5.40.1 # fetches & extracts 10-K text from SEC EDGAR
|
| 23 |
+
|
| 24 |
+
# --- Data loading (FinanceBench evaluation set) ---
|
| 25 |
+
# datasets must stay <3.0: FinanceBench-adjacent script datasets need it, and
|
| 26 |
# datasets>=3.0 removed support for script-based datasets.
|
| 27 |
datasets==2.21.0
|
| 28 |
pandas==3.0.3
|
src/config.py
CHANGED
|
@@ -18,30 +18,51 @@ DATA_DIR = PROJECT_ROOT / "data"
|
|
| 18 |
# writable, which makes chromadb's storage engine fail to start -> use
|
| 19 |
# /tmp, which is writable in any container.
|
| 20 |
# Override either default with the FINCHAT_VECTORSTORE env var.
|
|
|
|
| 21 |
_default_store = (
|
| 22 |
Path.home() / ".finchat" / "vectorstore"
|
| 23 |
if os.name == "nt"
|
| 24 |
else Path("/tmp/finchat/vectorstore")
|
| 25 |
)
|
| 26 |
-
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
#
|
| 30 |
-
#
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
# -
|
| 37 |
-
#
|
| 38 |
-
#
|
| 39 |
-
#
|
| 40 |
-
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
|
| 46 |
# --- Chunking ---------------------------------------------------------------
|
| 47 |
CHUNK_SIZE = 900 # characters per chunk
|
|
|
|
| 18 |
# writable, which makes chromadb's storage engine fail to start -> use
|
| 19 |
# /tmp, which is writable in any container.
|
| 20 |
# Override either default with the FINCHAT_VECTORSTORE env var.
|
| 21 |
+
_repo_store = PROJECT_ROOT / "vectorstore" # prebuilt index committed to repo
|
| 22 |
_default_store = (
|
| 23 |
Path.home() / ".finchat" / "vectorstore"
|
| 24 |
if os.name == "nt"
|
| 25 |
else Path("/tmp/finchat/vectorstore")
|
| 26 |
)
|
| 27 |
+
if os.getenv("FINCHAT_VECTORSTORE"):
|
| 28 |
+
VECTORSTORE_DIR = Path(os.environ["FINCHAT_VECTORSTORE"])
|
| 29 |
+
elif (_repo_store / "chroma.sqlite3").exists():
|
| 30 |
+
# A prebuilt index shipped with the repo (e.g. on Hugging Face Spaces) —
|
| 31 |
+
# use it directly so the app never rebuilds on startup.
|
| 32 |
+
VECTORSTORE_DIR = _repo_store
|
| 33 |
+
else:
|
| 34 |
+
VECTORSTORE_DIR = _default_store
|
| 35 |
+
|
| 36 |
+
# --- SEC EDGAR corpus -------------------------------------------------------
|
| 37 |
+
# Recent 10-K filings for recognizable companies, fetched from SEC EDGAR via
|
| 38 |
+
# edgartools. The set overlaps with the FinanceBench benchmark (matching
|
| 39 |
+
# company + fiscal year), so FinChat can be scored against it.
|
| 40 |
+
# SEC requires a contact identity (name/email) on every request.
|
| 41 |
+
EDGAR_IDENTITY = "narendra.daffa08@gmail.com"
|
| 42 |
+
|
| 43 |
+
# (ticker, display name, fiscal year). fiscal_year matches the filing's
|
| 44 |
+
# period_of_report year, which correctly handles offset fiscal years
|
| 45 |
+
# (e.g. Amcor closes in June, Nike in May).
|
| 46 |
+
TARGET_FILINGS = [
|
| 47 |
+
("AMD", "Advanced Micro Devices", 2022),
|
| 48 |
+
("AXP", "American Express", 2022),
|
| 49 |
+
("BA", "Boeing", 2022),
|
| 50 |
+
("PEP", "PepsiCo", 2022),
|
| 51 |
+
("AMCR", "Amcor", 2023),
|
| 52 |
+
("MMM", "3M", 2022),
|
| 53 |
+
("JNJ", "Johnson & Johnson", 2022),
|
| 54 |
+
("CVS", "CVS Health", 2022),
|
| 55 |
+
("PFE", "Pfizer", 2021),
|
| 56 |
+
("AES", "AES Corporation", 2022),
|
| 57 |
+
("VZ", "Verizon", 2022),
|
| 58 |
+
("BBY", "Best Buy", 2023),
|
| 59 |
+
("ADBE", "Adobe", 2022),
|
| 60 |
+
("ULTA", "Ulta Beauty", 2023),
|
| 61 |
+
("KO", "Coca-Cola", 2022),
|
| 62 |
+
("MSFT", "Microsoft", 2023),
|
| 63 |
+
("NKE", "Nike", 2023),
|
| 64 |
+
("GLW", "Corning", 2022),
|
| 65 |
+
]
|
| 66 |
|
| 67 |
# --- Chunking ---------------------------------------------------------------
|
| 68 |
CHUNK_SIZE = 900 # characters per chunk
|
src/ingest.py
CHANGED
|
@@ -1,17 +1,14 @@
|
|
| 1 |
-
"""Build the FinChat vector store from SEC 10-K filings.
|
| 2 |
|
| 3 |
Pipeline:
|
| 4 |
-
|
| 5 |
-
-> embed locally -> store in Chroma (persisted to .
|
| 6 |
|
| 7 |
-
Run
|
| 8 |
-
python -m src.ingest
|
| 9 |
-
python -m src.ingest --list # just list available companies, then exit
|
| 10 |
"""
|
| 11 |
from __future__ import annotations
|
| 12 |
|
| 13 |
-
import argparse
|
| 14 |
-
import re
|
| 15 |
import shutil
|
| 16 |
import sys
|
| 17 |
import time
|
|
@@ -20,8 +17,7 @@ from pathlib import Path
|
|
| 20 |
# Allow running as either `python -m src.ingest` or `python src/ingest.py`.
|
| 21 |
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
| 22 |
|
| 23 |
-
|
| 24 |
-
from datasets import load_dataset
|
| 25 |
from langchain_core.documents import Document
|
| 26 |
from langchain_text_splitters import RecursiveCharacterTextSplitter
|
| 27 |
from langchain_huggingface import HuggingFaceEmbeddings
|
|
@@ -29,105 +25,60 @@ from langchain_chroma import Chroma
|
|
| 29 |
|
| 30 |
from src import config
|
| 31 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
|
| 33 |
-
def pretty_section(label: str) -> str:
|
| 34 |
-
"""Turn a raw section label into a readable 'Section N' for display."""
|
| 35 |
-
nums = re.findall(r"\d+", str(label))
|
| 36 |
-
if nums:
|
| 37 |
-
return f"Section {nums[-1]}"
|
| 38 |
-
return str(label).replace("_", " ").strip().title() or "Filing"
|
| 39 |
|
|
|
|
|
|
|
| 40 |
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
|
| 49 |
|
| 50 |
-
# `tickers` is a list per row -> take the first as the primary ticker.
|
| 51 |
-
df["ticker"] = df["tickers"].apply(
|
| 52 |
-
lambda t: t[0] if hasattr(t, "__len__") and len(t) else None
|
| 53 |
-
)
|
| 54 |
-
# reportDate looks like "2020-09-26"; the year is the first 4 chars.
|
| 55 |
-
df["year"] = df["reportDate"].astype(str).str.slice(0, 4)
|
| 56 |
-
return df
|
| 57 |
-
|
| 58 |
-
|
| 59 |
-
def list_companies(df: pd.DataFrame, top: int = 30) -> None:
|
| 60 |
-
counts = (
|
| 61 |
-
df.dropna(subset=["ticker"])
|
| 62 |
-
.groupby(["ticker", "name"])
|
| 63 |
-
.size()
|
| 64 |
-
.sort_values(ascending=False)
|
| 65 |
-
.head(top)
|
| 66 |
-
)
|
| 67 |
-
print("\nTop companies available (ticker | name | #sentences):")
|
| 68 |
-
for (ticker, name), n in counts.items():
|
| 69 |
-
print(f" {ticker:<8} {str(name):<42} {n}")
|
| 70 |
-
print("\nCopy the tickers you want into TARGET_TICKERS in src/config.py.")
|
| 71 |
-
|
| 72 |
-
|
| 73 |
-
def select_rows(df: pd.DataFrame) -> pd.DataFrame:
|
| 74 |
-
df = df.dropna(subset=["ticker"]).copy()
|
| 75 |
-
|
| 76 |
-
if config.TARGET_YEARS:
|
| 77 |
-
years = {str(y) for y in config.TARGET_YEARS}
|
| 78 |
-
df = df[df["year"].isin(years)]
|
| 79 |
-
|
| 80 |
-
if config.TARGET_TICKERS:
|
| 81 |
-
wanted = {t.upper() for t in config.TARGET_TICKERS}
|
| 82 |
-
df = df[df["ticker"].str.upper().isin(wanted)]
|
| 83 |
-
else:
|
| 84 |
-
# No explicit list -> keep the TOP_N companies by content volume.
|
| 85 |
-
top = (
|
| 86 |
-
df.groupby("ticker").size()
|
| 87 |
-
.sort_values(ascending=False)
|
| 88 |
-
.head(config.TOP_N_COMPANIES)
|
| 89 |
-
.index
|
| 90 |
-
)
|
| 91 |
-
df = df[df["ticker"].isin(top)]
|
| 92 |
-
|
| 93 |
-
return df
|
| 94 |
-
|
| 95 |
-
|
| 96 |
-
def build_documents(df: pd.DataFrame) -> list[Document]:
|
| 97 |
-
"""Reassemble sentences into section text, then split into chunks."""
|
| 98 |
-
# One "raw document" per (filing, section). docID identifies one filing.
|
| 99 |
-
grouped = df.sort_values("sentenceCount").groupby(["docID", "section"])
|
| 100 |
-
|
| 101 |
-
raw_docs: list[Document] = []
|
| 102 |
-
for (doc_id, section), rows in grouped:
|
| 103 |
-
text = " ".join(str(s) for s in rows["sentence"].tolist()).strip()
|
| 104 |
-
if len(text) < 50: # skip near-empty sections
|
| 105 |
-
continue
|
| 106 |
-
head = rows.iloc[0]
|
| 107 |
-
company = str(head["name"])
|
| 108 |
-
year = str(head["year"])
|
| 109 |
-
raw_docs.append(
|
| 110 |
-
Document(
|
| 111 |
-
page_content=text,
|
| 112 |
-
metadata={
|
| 113 |
-
"ticker": str(head["ticker"]),
|
| 114 |
-
"company": company,
|
| 115 |
-
"year": year,
|
| 116 |
-
"section": str(section),
|
| 117 |
-
"cik": str(head["cik"]),
|
| 118 |
-
"docID": str(doc_id),
|
| 119 |
-
"source": f"{company} 10-K ({year}) - {pretty_section(section)}",
|
| 120 |
-
},
|
| 121 |
-
)
|
| 122 |
-
)
|
| 123 |
|
|
|
|
|
|
|
|
|
|
| 124 |
splitter = RecursiveCharacterTextSplitter(
|
| 125 |
chunk_size=config.CHUNK_SIZE,
|
| 126 |
chunk_overlap=config.CHUNK_OVERLAP,
|
| 127 |
)
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 131 |
|
| 132 |
|
| 133 |
def _safe_rmtree(path: Path, retries: int = 3) -> None:
|
|
@@ -138,7 +89,7 @@ def _safe_rmtree(path: Path, retries: int = 3) -> None:
|
|
| 138 |
return
|
| 139 |
except (PermissionError, OSError):
|
| 140 |
time.sleep(1.0)
|
| 141 |
-
shutil.rmtree(path)
|
| 142 |
|
| 143 |
|
| 144 |
def build_vectorstore(chunks: list[Document]) -> None:
|
|
@@ -150,7 +101,7 @@ def build_vectorstore(chunks: list[Document]) -> None:
|
|
| 150 |
print(f"Loading embedding model {config.EMBEDDING_MODEL} ...")
|
| 151 |
embeddings = HuggingFaceEmbeddings(model_name=config.EMBEDDING_MODEL)
|
| 152 |
|
| 153 |
-
print("Embedding & storing chunks
|
| 154 |
Chroma.from_documents(
|
| 155 |
documents=chunks,
|
| 156 |
embedding=embeddings,
|
|
@@ -161,34 +112,18 @@ def build_vectorstore(chunks: list[Document]) -> None:
|
|
| 161 |
|
| 162 |
|
| 163 |
def build_index() -> None:
|
| 164 |
-
"""
|
| 165 |
|
| 166 |
-
Importable so the app can bootstrap the
|
| 167 |
-
(e.g. on a fresh Hugging Face Space).
|
| 168 |
"""
|
| 169 |
-
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
"Run `python -m src.ingest --list` to see what's available."
|
| 175 |
-
)
|
| 176 |
-
companies = sorted(selected["ticker"].unique())
|
| 177 |
-
print(f"Ingesting {len(companies)} companies: {', '.join(companies)}")
|
| 178 |
-
chunks = build_documents(selected)
|
| 179 |
-
build_vectorstore(chunks)
|
| 180 |
|
| 181 |
|
| 182 |
def main() -> None:
|
| 183 |
-
parser = argparse.ArgumentParser(description="Build the FinChat vector store.")
|
| 184 |
-
parser.add_argument("--list", action="store_true",
|
| 185 |
-
help="List available companies and exit.")
|
| 186 |
-
args = parser.parse_args()
|
| 187 |
-
|
| 188 |
-
if args.list:
|
| 189 |
-
list_companies(load_dataframe())
|
| 190 |
-
return
|
| 191 |
-
|
| 192 |
build_index()
|
| 193 |
|
| 194 |
|
|
|
|
| 1 |
+
"""Build the FinChat vector store from recent SEC 10-K filings (EDGAR).
|
| 2 |
|
| 3 |
Pipeline:
|
| 4 |
+
fetch each target 10-K (edgartools) -> split into chunks
|
| 5 |
+
-> embed locally -> store in Chroma (persisted to config.VECTORSTORE_DIR)
|
| 6 |
|
| 7 |
+
Run once, from the project root:
|
| 8 |
+
python -m src.ingest
|
|
|
|
| 9 |
"""
|
| 10 |
from __future__ import annotations
|
| 11 |
|
|
|
|
|
|
|
| 12 |
import shutil
|
| 13 |
import sys
|
| 14 |
import time
|
|
|
|
| 17 |
# Allow running as either `python -m src.ingest` or `python src/ingest.py`.
|
| 18 |
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
| 19 |
|
| 20 |
+
from edgar import Company, set_identity
|
|
|
|
| 21 |
from langchain_core.documents import Document
|
| 22 |
from langchain_text_splitters import RecursiveCharacterTextSplitter
|
| 23 |
from langchain_huggingface import HuggingFaceEmbeddings
|
|
|
|
| 25 |
|
| 26 |
from src import config
|
| 27 |
|
| 28 |
+
# Windows consoles default to cp1252 and crash when print() emits Unicode
|
| 29 |
+
# (arrows, em-dashes, curly quotes from filings). Force UTF-8 output.
|
| 30 |
+
try:
|
| 31 |
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
| 32 |
+
except Exception:
|
| 33 |
+
pass
|
| 34 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
|
| 36 |
+
def fetch_filing(ticker: str, fiscal_year: int) -> tuple[str, str] | None:
|
| 37 |
+
"""Return (text, accession_no) for the 10-K whose fiscal period matches.
|
| 38 |
|
| 39 |
+
Matches on the filing's period_of_report year, so an offset fiscal year
|
| 40 |
+
(e.g. Amcor's June close) still resolves to the right filing.
|
| 41 |
+
"""
|
| 42 |
+
for f in Company(ticker).get_filings(form="10-K"):
|
| 43 |
+
period = getattr(f, "period_of_report", None)
|
| 44 |
+
if period and str(period)[:4] == str(fiscal_year):
|
| 45 |
+
return f.text(), f.accession_no
|
| 46 |
+
return None
|
| 47 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
|
| 49 |
+
def fetch_documents() -> list[Document]:
|
| 50 |
+
"""Fetch every target filing and split it into chunked LangChain Documents."""
|
| 51 |
+
set_identity(config.EDGAR_IDENTITY)
|
| 52 |
splitter = RecursiveCharacterTextSplitter(
|
| 53 |
chunk_size=config.CHUNK_SIZE,
|
| 54 |
chunk_overlap=config.CHUNK_OVERLAP,
|
| 55 |
)
|
| 56 |
+
|
| 57 |
+
docs: list[Document] = []
|
| 58 |
+
for ticker, name, year in config.TARGET_FILINGS:
|
| 59 |
+
print(f"Fetching {ticker} FY{year} 10-K ...", end=" ", flush=True)
|
| 60 |
+
result = fetch_filing(ticker, year)
|
| 61 |
+
if result is None:
|
| 62 |
+
print("NOT FOUND -- skipping")
|
| 63 |
+
continue
|
| 64 |
+
text, accession = result
|
| 65 |
+
source = f"{name} 10-K (FY{year})"
|
| 66 |
+
for chunk in splitter.split_text(text):
|
| 67 |
+
docs.append(
|
| 68 |
+
Document(
|
| 69 |
+
page_content=chunk,
|
| 70 |
+
metadata={
|
| 71 |
+
"ticker": ticker,
|
| 72 |
+
"company": name,
|
| 73 |
+
"year": str(year),
|
| 74 |
+
"accession": accession,
|
| 75 |
+
"source": source,
|
| 76 |
+
},
|
| 77 |
+
)
|
| 78 |
+
)
|
| 79 |
+
print(f"{len(text):,} chars -> {len(docs)} chunks so far")
|
| 80 |
+
time.sleep(0.5) # be polite to SEC's servers
|
| 81 |
+
return docs
|
| 82 |
|
| 83 |
|
| 84 |
def _safe_rmtree(path: Path, retries: int = 3) -> None:
|
|
|
|
| 89 |
return
|
| 90 |
except (PermissionError, OSError):
|
| 91 |
time.sleep(1.0)
|
| 92 |
+
shutil.rmtree(path)
|
| 93 |
|
| 94 |
|
| 95 |
def build_vectorstore(chunks: list[Document]) -> None:
|
|
|
|
| 101 |
print(f"Loading embedding model {config.EMBEDDING_MODEL} ...")
|
| 102 |
embeddings = HuggingFaceEmbeddings(model_name=config.EMBEDDING_MODEL)
|
| 103 |
|
| 104 |
+
print(f"Embedding & storing {len(chunks)} chunks (a few minutes) ...")
|
| 105 |
Chroma.from_documents(
|
| 106 |
documents=chunks,
|
| 107 |
embedding=embeddings,
|
|
|
|
| 112 |
|
| 113 |
|
| 114 |
def build_index() -> None:
|
| 115 |
+
"""Full pipeline: fetch filings -> chunk -> embed -> store.
|
| 116 |
|
| 117 |
+
Importable so the app can bootstrap the store on first run.
|
|
|
|
| 118 |
"""
|
| 119 |
+
docs = fetch_documents()
|
| 120 |
+
if not docs:
|
| 121 |
+
raise SystemExit("No filings were fetched — check tickers/years in config.py.")
|
| 122 |
+
print(f"Total: {len(docs)} chunks from {len(config.TARGET_FILINGS)} target filings.")
|
| 123 |
+
build_vectorstore(docs)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 124 |
|
| 125 |
|
| 126 |
def main() -> None:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 127 |
build_index()
|
| 128 |
|
| 129 |
|
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/data_level0.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c6b529e7f484c17ad5a211753037d3fd28fd20e1ccbd9f9358289b6a69c4be31
|
| 3 |
+
size 27018796
|
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/header.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bcdaa3a4152103e1edc94b56aad83a3557a6828134c42d68f9260717e1f9f243
|
| 3 |
+
size 100
|
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/index_metadata.pickle
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:54c22b54ab9f504cadfc301b506ca9f2b0b7769736bc624af5fa2402d09716cc
|
| 3 |
+
size 1483324
|
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/length.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:034a92141994e5554a9ca608b1f904f4c4a35b65799fffad10c0086c0505943e
|
| 3 |
+
size 64484
|
vectorstore/5e736a2d-5524-4786-879f-5e5bfddfc2b6/link_lists.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2ddeffe7ec74e3fc7fbdf981a4b7e3d733cdcaabe2aa54b6446ccd870654afe9
|
| 3 |
+
size 139012
|
vectorstore/chroma.sqlite3
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ab762371a44ad19ed423c7b6c4d3b94432832e17dca56a2cab388d02882ebe1
|
| 3 |
+
size 99049472
|