Spaces:
Running
Running
Deploy fail-closed helper 436d652
Browse filesRefresh live offer evidence, validate exact retrieved claims, and hand unsupported questions to the canonical contact form.
- .env.example +17 -4
- .github/workflows/ci.yml +1 -0
- .github/workflows/keepalive.yml +1 -1
- README.md +77 -9
- data/assistant_notes.json +47 -20
- data/discovered_urls.json +34 -1
- data/pages.json +0 -0
- data/towardsai_com_pages.json +0 -0
- scripts/build_towardsai_com_catalog.py +1342 -0
- static/widget.js +33 -6
- tai_helper/api.py +158 -26
- tai_helper/catalog.py +1489 -73
- tai_helper/llm.py +1168 -42
- tai_helper/monitoring.py +6 -4
- tai_helper/schemas.py +3 -0
- tai_helper/settings.py +82 -10
- tests/conftest.py +14 -0
- tests/test_api.py +537 -16
- tests/test_catalog.py +187 -3
- tests/test_catalog_builder.py +487 -0
- tests/test_grounding.py +866 -0
- tests/test_live_smoke.py +66 -4
- tests/test_llm.py +189 -0
- tests/test_offers.py +753 -0
- tests/test_retrieval_safety.py +414 -0
- tests/test_widget.py +14 -0
.env.example
CHANGED
|
@@ -1,11 +1,21 @@
|
|
| 1 |
# Public domains where the helper widget is allowed to run.
|
| 2 |
-
HELPER_ALLOWED_ORIGINS=https://academy.towardsai.net,https://towardsai.net,https://www.towardsai.net
|
| 3 |
-
HELPER_ALLOWED_HOSTS=academy.towardsai.net,towardsai.net,www.towardsai.net
|
| 4 |
|
| 5 |
-
#
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
GEMINI_API_KEY=
|
| 7 |
-
|
| 8 |
HELPER_MAX_OUTPUT_TOKENS=420
|
|
|
|
| 9 |
|
| 10 |
# Hard public rate limits.
|
| 11 |
HELPER_RATE_LIMIT_PER_MINUTE=3
|
|
@@ -17,6 +27,9 @@ HELPER_MAX_BODY_BYTES=65536
|
|
| 17 |
HELPER_MAX_QUERY_CHARS=600
|
| 18 |
HELPER_MAX_HISTORY_TURNS=8
|
| 19 |
|
|
|
|
|
|
|
|
|
|
| 20 |
# Optional Opik monitoring.
|
| 21 |
OPIK_ENABLED=false
|
| 22 |
OPIK_PROJECT_NAME=towards-ai-helper
|
|
|
|
| 1 |
# Public domains where the helper widget is allowed to run.
|
| 2 |
+
HELPER_ALLOWED_ORIGINS=https://towardsai.com,https://www.towardsai.com,https://academy.towardsai.net,https://towardsai.net,https://www.towardsai.net
|
| 3 |
+
HELPER_ALLOWED_HOSTS=towardsai.com,www.towardsai.com,academy.towardsai.net,towardsai.net,www.towardsai.net
|
| 4 |
|
| 5 |
+
# These public hosts allow the widget on every path except private/admin routes.
|
| 6 |
+
HELPER_SITE_WIDE_HOSTS=towardsai.com,www.towardsai.com
|
| 7 |
+
|
| 8 |
+
# Primary model: DeepSeek V4 Flash through the DeepSeek API.
|
| 9 |
+
DEEPSEEK_API_KEY=
|
| 10 |
+
DEEPSEEK_BASE_URL=https://api.deepseek.com
|
| 11 |
+
HELPER_PRIMARY_MODEL=deepseek-v4-flash
|
| 12 |
+
HELPER_DEEPSEEK_THINKING=disabled
|
| 13 |
+
|
| 14 |
+
# Backup model: Gemini 2.5 Flash.
|
| 15 |
GEMINI_API_KEY=
|
| 16 |
+
HELPER_FALLBACK_MODEL=gemini-2.5-flash
|
| 17 |
HELPER_MAX_OUTPUT_TOKENS=420
|
| 18 |
+
HELPER_LLM_REQUEST_TIMEOUT_SECONDS=20
|
| 19 |
|
| 20 |
# Hard public rate limits.
|
| 21 |
HELPER_RATE_LIMIT_PER_MINUTE=3
|
|
|
|
| 27 |
HELPER_MAX_QUERY_CHARS=600
|
| 28 |
HELPER_MAX_HISTORY_TURNS=8
|
| 29 |
|
| 30 |
+
# Fail closed when the local public-page catalog has not been refreshed recently.
|
| 31 |
+
HELPER_CATALOG_MAX_AGE_DAYS=14
|
| 32 |
+
|
| 33 |
# Optional Opik monitoring.
|
| 34 |
OPIK_ENABLED=false
|
| 35 |
OPIK_PROJECT_NAME=towards-ai-helper
|
.github/workflows/ci.yml
CHANGED
|
@@ -55,6 +55,7 @@ jobs:
|
|
| 55 |
|
| 56 |
env:
|
| 57 |
LIVE_SPACE_BASE_URL: ${{ vars.LIVE_SPACE_BASE_URL || 'https://towardsai-tutors-tai-helper.hf.space' }}
|
|
|
|
| 58 |
RUN_LIVE_CHAT_SMOKE: ${{ github.event.inputs.run_live_chat || 'false' }}
|
| 59 |
|
| 60 |
steps:
|
|
|
|
| 55 |
|
| 56 |
env:
|
| 57 |
LIVE_SPACE_BASE_URL: ${{ vars.LIVE_SPACE_BASE_URL || 'https://towardsai-tutors-tai-helper.hf.space' }}
|
| 58 |
+
LIVE_HELPER_WIDGET_PATH: ${{ vars.LIVE_HELPER_WIDGET_PATH || '/helper-widget.js' }}
|
| 59 |
RUN_LIVE_CHAT_SMOKE: ${{ github.event.inputs.run_live_chat || 'false' }}
|
| 60 |
|
| 61 |
steps:
|
.github/workflows/keepalive.yml
CHANGED
|
@@ -21,4 +21,4 @@ jobs:
|
|
| 21 |
|
| 22 |
- name: Widget check
|
| 23 |
run: |
|
| 24 |
-
curl --fail --show-error --silent --max-time 60 "$SPACE_BASE_URL/widget.js" | grep -q "Towards AI Helper"
|
|
|
|
| 21 |
|
| 22 |
- name: Widget check
|
| 23 |
run: |
|
| 24 |
+
curl --fail --show-error --silent --max-time 60 "$SPACE_BASE_URL/helper-widget.js" | grep -q "Towards AI Helper"
|
README.md
CHANGED
|
@@ -7,35 +7,49 @@ pinned: false
|
|
| 7 |
|
| 8 |
# Towards AI Helper
|
| 9 |
|
| 10 |
-
Public sales helper for anonymous visitors on
|
| 11 |
-
|
| 12 |
-
mentorship, free resources, the book, or B2B
|
|
|
|
| 13 |
|
| 14 |
It is deliberately separate from the Thinkific lesson tutor:
|
| 15 |
|
| 16 |
-
- This helper appears
|
|
|
|
| 17 |
- It hides when a visitor appears signed in.
|
| 18 |
- The first message must be one of the fixed prompt buttons.
|
| 19 |
-
- It uses
|
|
|
|
| 20 |
- It does not answer general AI questions or give away course lesson content.
|
|
|
|
|
|
|
|
|
|
| 21 |
|
| 22 |
## Local Setup
|
| 23 |
|
| 24 |
```bash
|
| 25 |
-
cd /
|
| 26 |
uv sync
|
| 27 |
cp .env.example .env
|
| 28 |
uv run uvicorn tai_helper.api:app --host 0.0.0.0 --port 8001
|
| 29 |
```
|
| 30 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 31 |
## Public Widget Snippet
|
| 32 |
|
| 33 |
-
Add this to
|
| 34 |
-
`towardsai.net`
|
|
|
|
| 35 |
|
| 36 |
```html
|
| 37 |
<script
|
| 38 |
-
src="https://YOUR-HF-SPACE/widget.js"
|
| 39 |
data-api-base="https://YOUR-HF-SPACE"
|
| 40 |
defer
|
| 41 |
></script>
|
|
@@ -44,6 +58,60 @@ Add this to public site footer code on `academy.towardsai.net` and
|
|
| 44 |
The widget fetches `/api/helper/config`, checks the current public URL, hides on
|
| 45 |
signed-in sessions, and starts as a bottom-right `Ask the helper` bubble.
|
| 46 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 47 |
## Monitoring
|
| 48 |
|
| 49 |
Enable Opik:
|
|
|
|
| 7 |
|
| 8 |
# Towards AI Helper
|
| 9 |
|
| 10 |
+
Public sales helper for anonymous visitors on `towardsai.com`, the legacy
|
| 11 |
+
`towardsai.net` domain, and Towards AI Academy. It helps prospective students
|
| 12 |
+
choose courses, bundles, mentorship, free resources, the book, or B2B
|
| 13 |
+
training/consulting.
|
| 14 |
|
| 15 |
It is deliberately separate from the Thinkific lesson tutor:
|
| 16 |
|
| 17 |
+
- This helper appears on public `towardsai.com` pages and selected public
|
| 18 |
+
Academy/legacy `.net` pages.
|
| 19 |
- It hides when a visitor appears signed in.
|
| 20 |
- The first message must be one of the fixed prompt buttons.
|
| 21 |
+
- It uses DeepSeek V4 Flash through the DeepSeek API first, then falls back to
|
| 22 |
+
Gemini 2.5 Flash if the primary provider fails.
|
| 23 |
- It does not answer general AI questions or give away course lesson content.
|
| 24 |
+
- It returns offer facts only after every sentence has passed exact-quote and
|
| 25 |
+
retrieved-chunk validation. Missing, stale, conflicting, or invalid evidence
|
| 26 |
+
produces an explicit “I couldn't verify that” response instead of a guess.
|
| 27 |
|
| 28 |
## Local Setup
|
| 29 |
|
| 30 |
```bash
|
| 31 |
+
cd /path/to/tai-helper
|
| 32 |
uv sync
|
| 33 |
cp .env.example .env
|
| 34 |
uv run uvicorn tai_helper.api:app --host 0.0.0.0 --port 8001
|
| 35 |
```
|
| 36 |
|
| 37 |
+
Required model secrets:
|
| 38 |
+
|
| 39 |
+
```bash
|
| 40 |
+
DEEPSEEK_API_KEY=...
|
| 41 |
+
GEMINI_API_KEY=...
|
| 42 |
+
```
|
| 43 |
+
|
| 44 |
## Public Widget Snippet
|
| 45 |
|
| 46 |
+
Add this to the global footer code on `towardsai.com`. Add it separately to
|
| 47 |
+
`academy.towardsai.net` if the helper should also appear on its signed-out
|
| 48 |
+
public pages:
|
| 49 |
|
| 50 |
```html
|
| 51 |
<script
|
| 52 |
+
src="https://YOUR-HF-SPACE/helper-widget.js"
|
| 53 |
data-api-base="https://YOUR-HF-SPACE"
|
| 54 |
defer
|
| 55 |
></script>
|
|
|
|
| 58 |
The widget fetches `/api/helper/config`, checks the current public URL, hides on
|
| 59 |
signed-in sessions, and starts as a bottom-right `Ask the helper` bubble.
|
| 60 |
|
| 61 |
+
The `.net` website redirects to `.com` before page JavaScript runs, so installing
|
| 62 |
+
the snippet only on `.net` will not make the helper appear on `.com`.
|
| 63 |
+
|
| 64 |
+
## Deployment Domain Settings
|
| 65 |
+
|
| 66 |
+
The defaults include both `.com` and `.net`. If the deployed Space already has
|
| 67 |
+
these environment variables set, update them because deployed values override
|
| 68 |
+
the defaults:
|
| 69 |
+
|
| 70 |
+
```bash
|
| 71 |
+
HELPER_ALLOWED_ORIGINS=https://towardsai.com,https://www.towardsai.com,https://academy.towardsai.net,https://towardsai.net,https://www.towardsai.net
|
| 72 |
+
HELPER_ALLOWED_HOSTS=towardsai.com,www.towardsai.com,academy.towardsai.net,towardsai.net,www.towardsai.net
|
| 73 |
+
HELPER_SITE_WIDE_HOSTS=towardsai.com,www.towardsai.com
|
| 74 |
+
```
|
| 75 |
+
|
| 76 |
+
`HELPER_SITE_WIDE_HOSTS` makes the widget available on current and future public
|
| 77 |
+
`.com` paths. Signed-in sessions, checkout/account paths, WordPress admin paths,
|
| 78 |
+
API paths, and previews remain blocked.
|
| 79 |
+
|
| 80 |
+
## Knowledge Catalog
|
| 81 |
+
|
| 82 |
+
`data/towardsai_com_pages.json` is generated from every URL in the current
|
| 83 |
+
`towardsai.com` page sitemap. `data/pages.json` is generated from every public
|
| 84 |
+
Academy sitemap URL. Each eligible page records its canonical URL, successful
|
| 85 |
+
fetch time, content hash, source authority, heading-aware chunks, and atomic
|
| 86 |
+
evidence spans (complete source sentences, DOM blocks, or table rows). A second
|
| 87 |
+
hash binds the canonical URL, page text, chunks, and span definitions together.
|
| 88 |
+
Known staging pages, legacy duplicates, conflicting catalog summaries, failed
|
| 89 |
+
fetches, and manually described links remain in the inventory but are explicitly
|
| 90 |
+
ineligible as evidence.
|
| 91 |
+
|
| 92 |
+
Refresh both catalogs after public pages change:
|
| 93 |
+
|
| 94 |
+
```bash
|
| 95 |
+
python scripts/build_towardsai_com_catalog.py
|
| 96 |
+
```
|
| 97 |
+
|
| 98 |
+
The API accepts evidence only while each page's successful fetch is within
|
| 99 |
+
`HELPER_CATALOG_MAX_AGE_DAYS` (14 days by default). Canonical `.com` offer pages
|
| 100 |
+
supersede lower-authority Academy mirrors for the same offer. Routing notes are
|
| 101 |
+
never treated as factual evidence.
|
| 102 |
+
|
| 103 |
+
For factual answers, the model must return structured sentence-level claims.
|
| 104 |
+
Every claim needs a valid retrieved chunk ID and must copy one complete
|
| 105 |
+
server-defined evidence span. Arbitrary substrings are forbidden, so a model
|
| 106 |
+
cannot turn “No code required” into “code required” or cross table-row
|
| 107 |
+
boundaries. The server independently verifies catalog freshness, both hashes,
|
| 108 |
+
canonical URLs, chunk/span identity, numbers, prices, percentages, URLs,
|
| 109 |
+
negation, the named offer, the requested fact type, and output schema before
|
| 110 |
+
anything is shown to a visitor. If exact target-qualified evidence is missing or
|
| 111 |
+
generation fails validation, the API returns no sources and directs the visitor
|
| 112 |
+
to the canonical [Towards AI contact form](https://towardsai.com/academy/contact/#contact)
|
| 113 |
+
instead of guessing.
|
| 114 |
+
|
| 115 |
## Monitoring
|
| 116 |
|
| 117 |
Enable Opik:
|
data/assistant_notes.json
CHANGED
|
@@ -15,50 +15,75 @@
|
|
| 15 |
"Keep answers concise and useful to control costs.",
|
| 16 |
"For B2B, company training, and consulting leads, ask for company/team context and recommend emailing louis@towardsai.net.",
|
| 17 |
"For coupon code requests, do not provide a code. If the user insists, tell them to email louis@towardsai.net with context and what they need.",
|
| 18 |
-
"For eager learners who want broad access, recommend the Get it all bundle as the strongest value option."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
],
|
| 20 |
"course_routing": [
|
| 21 |
{
|
| 22 |
-
"name": "Get
|
| 23 |
-
"url": "https://
|
| 24 |
"when_to_recommend": "Best value for motivated learners who want broad access across the academy or are serious about going from fundamentals to advanced AI engineering."
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"name": "Python for
|
| 28 |
-
"url": "https://
|
| 29 |
"when_to_recommend": "Best first step for beginners, non-coders, or people who need Python foundations before building LLM products."
|
| 30 |
},
|
| 31 |
{
|
| 32 |
-
"name": "Full Stack AI Engineering
|
| 33 |
-
"url": "https://
|
| 34 |
"when_to_recommend": "Best for builders who want to ship LLM products with RAG, context engineering, deployment, and product-oriented AI engineering."
|
| 35 |
},
|
| 36 |
{
|
| 37 |
-
"name": "Agent Engineering
|
| 38 |
-
"url": "https://
|
| 39 |
"when_to_recommend": "Best for intermediate to advanced engineers who know Python/APIs and want production-grade autonomous agents, evaluation, monitoring, and deployment."
|
| 40 |
},
|
| 41 |
{
|
| 42 |
-
"name": "
|
| 43 |
-
"url": "https://
|
| 44 |
"when_to_recommend": "Best for business users, operators, founders, and professionals who want practical AI workflows without heavy coding."
|
| 45 |
},
|
| 46 |
{
|
| 47 |
"name": "10-Hours LLM Fundamentals",
|
| 48 |
-
"url": "https://
|
| 49 |
"when_to_recommend": "Best quick primer for LLM fundamentals before taking a deeper course."
|
| 50 |
},
|
| 51 |
{
|
| 52 |
"name": "Building LLMs for Production",
|
| 53 |
-
"url": "https://
|
| 54 |
"when_to_recommend": "Best for learners who want production LLM engineering concepts connected to the Towards AI book/resources."
|
| 55 |
},
|
| 56 |
{
|
| 57 |
"name": "Towards AI Mentorship",
|
| 58 |
-
"url": "https://
|
| 59 |
"when_to_recommend": "Best when the user asks for mentors, personalized guidance, accountability, project feedback, or career/course direction."
|
| 60 |
}
|
| 61 |
],
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
"external_resources": [
|
| 63 |
{
|
| 64 |
"name": "What's AI YouTube",
|
|
@@ -67,12 +92,12 @@
|
|
| 67 |
},
|
| 68 |
{
|
| 69 |
"name": "Towards AI Resource Library",
|
| 70 |
-
"url": "https://towardsai.
|
| 71 |
"when_to_recommend": "Free articles, guides, and learning resources."
|
| 72 |
},
|
| 73 |
{
|
| 74 |
"name": "Towards AI book resources",
|
| 75 |
-
"url": "https://towardsai.
|
| 76 |
"when_to_recommend": "Book companion resources and production LLM learning path."
|
| 77 |
},
|
| 78 |
{
|
|
@@ -83,10 +108,12 @@
|
|
| 83 |
],
|
| 84 |
"b2b": {
|
| 85 |
"urls": [
|
| 86 |
-
"https://towardsai.
|
| 87 |
-
"https://towardsai.
|
|
|
|
|
|
|
| 88 |
],
|
| 89 |
"email": "louis@towardsai.net",
|
| 90 |
-
"instruction": "Ask for company size, team background, use cases, desired
|
| 91 |
}
|
| 92 |
-
}
|
|
|
|
| 15 |
"Keep answers concise and useful to control costs.",
|
| 16 |
"For B2B, company training, and consulting leads, ask for company/team context and recommend emailing louis@towardsai.net.",
|
| 17 |
"For coupon code requests, do not provide a code. If the user insists, tell them to email louis@towardsai.net with context and what they need.",
|
| 18 |
+
"For eager learners who want broad access, recommend the Get it all bundle as the strongest value option.",
|
| 19 |
+
"Never state an exact price, discount, product count, included course, access duration, refund term, or other offer detail unless that fact is explicitly present in the retrieved source text. If it is not present, say you cannot confirm it from the current public pages and link to the relevant page or contact."
|
| 20 |
+
],
|
| 21 |
+
"catalog_hubs": [
|
| 22 |
+
{
|
| 23 |
+
"name": "Towards AI Academy",
|
| 24 |
+
"url": "https://towardsai.com/academy/",
|
| 25 |
+
"when_to_recommend": "Use this as the central overview of Towards AI learning options. For current inclusions, prices, discounts, and terms, rely only on exact text retrieved from the relevant offer page."
|
| 26 |
+
}
|
| 27 |
],
|
| 28 |
"course_routing": [
|
| 29 |
{
|
| 30 |
+
"name": "Get It All: Every Course, One Bundle",
|
| 31 |
+
"url": "https://towardsai.com/academy/bundles/get-it-all/",
|
| 32 |
"when_to_recommend": "Best value for motivated learners who want broad access across the academy or are serious about going from fundamentals to advanced AI engineering."
|
| 33 |
},
|
| 34 |
{
|
| 35 |
+
"name": "Beginner Python for AI Engineering",
|
| 36 |
+
"url": "https://towardsai.com/academy/python-for-ai-engineering/",
|
| 37 |
"when_to_recommend": "Best first step for beginners, non-coders, or people who need Python foundations before building LLM products."
|
| 38 |
},
|
| 39 |
{
|
| 40 |
+
"name": "Full Stack AI Engineering",
|
| 41 |
+
"url": "https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 42 |
"when_to_recommend": "Best for builders who want to ship LLM products with RAG, context engineering, deployment, and product-oriented AI engineering."
|
| 43 |
},
|
| 44 |
{
|
| 45 |
+
"name": "Agent Engineering",
|
| 46 |
+
"url": "https://towardsai.com/academy/agent-engineering/",
|
| 47 |
"when_to_recommend": "Best for intermediate to advanced engineers who know Python/APIs and want production-grade autonomous agents, evaluation, monitoring, and deployment."
|
| 48 |
},
|
| 49 |
{
|
| 50 |
+
"name": "Master AI for Work",
|
| 51 |
+
"url": "https://towardsai.com/academy/ai-for-work/",
|
| 52 |
"when_to_recommend": "Best for business users, operators, founders, and professionals who want practical AI workflows without heavy coding."
|
| 53 |
},
|
| 54 |
{
|
| 55 |
"name": "10-Hours LLM Fundamentals",
|
| 56 |
+
"url": "https://towardsai.com/academy/llm-primer/",
|
| 57 |
"when_to_recommend": "Best quick primer for LLM fundamentals before taking a deeper course."
|
| 58 |
},
|
| 59 |
{
|
| 60 |
"name": "Building LLMs for Production",
|
| 61 |
+
"url": "https://towardsai.com/academy/building-llms-for-production/",
|
| 62 |
"when_to_recommend": "Best for learners who want production LLM engineering concepts connected to the Towards AI book/resources."
|
| 63 |
},
|
| 64 |
{
|
| 65 |
"name": "Towards AI Mentorship",
|
| 66 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 67 |
"when_to_recommend": "Best when the user asks for mentors, personalized guidance, accountability, project feedback, or career/course direction."
|
| 68 |
}
|
| 69 |
],
|
| 70 |
+
"bundle_routing": [
|
| 71 |
+
{
|
| 72 |
+
"name": "From Developer to Advanced AI Engineer",
|
| 73 |
+
"url": "https://towardsai.com/academy/bundles/10-hour-crash-course-into-llm-developer-expert/",
|
| 74 |
+
"when_to_recommend": "Best for developers who already code and want a path from LLM fundamentals into advanced AI engineering."
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"name": "From Non-Coder to AI Engineer",
|
| 78 |
+
"url": "https://towardsai.com/academy/bundles/from-coding-novice-to-advanced-llm-developer/",
|
| 79 |
+
"when_to_recommend": "Best for learners who need Python foundations before progressing through the AI engineering curriculum."
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"name": "Get It All",
|
| 83 |
+
"url": "https://towardsai.com/academy/bundles/get-it-all/",
|
| 84 |
+
"when_to_recommend": "Best-value choice for learners who want every course and the broadest path from novice to expert."
|
| 85 |
+
}
|
| 86 |
+
],
|
| 87 |
"external_resources": [
|
| 88 |
{
|
| 89 |
"name": "What's AI YouTube",
|
|
|
|
| 92 |
},
|
| 93 |
{
|
| 94 |
"name": "Towards AI Resource Library",
|
| 95 |
+
"url": "https://towardsai.com/towards-ai-resource-library",
|
| 96 |
"when_to_recommend": "Free articles, guides, and learning resources."
|
| 97 |
},
|
| 98 |
{
|
| 99 |
"name": "Towards AI book resources",
|
| 100 |
+
"url": "https://towardsai.com/academy/book/",
|
| 101 |
"when_to_recommend": "Book companion resources and production LLM learning path."
|
| 102 |
},
|
| 103 |
{
|
|
|
|
| 108 |
],
|
| 109 |
"b2b": {
|
| 110 |
"urls": [
|
| 111 |
+
"https://towardsai.com/enterpriseenablement/",
|
| 112 |
+
"https://towardsai.com/valuecreation/",
|
| 113 |
+
"https://towardsai.com/enterprise/software-developer-to-ai-engineer/",
|
| 114 |
+
"https://towardsai.com/enterprise/agentic-developer-conversion/"
|
| 115 |
],
|
| 116 |
"email": "louis@towardsai.net",
|
| 117 |
+
"instruction": "Ask for company size, team background, use cases, desired outcome, and timeline. Route software developer conversion to the AI Engineer programme; Claude Code, Codex, or coding-agent adoption to Agentic Developer Conversion; broader deployment or custom development to Value Creation; and general team training to Enterprise Enablement. Suggest emailing louis@towardsai.net with details or continuing in chat."
|
| 118 |
}
|
| 119 |
+
}
|
data/discovered_urls.json
CHANGED
|
@@ -1,6 +1,7 @@
|
|
| 1 |
{
|
| 2 |
"academy_sitemap_url": "https://academy.towardsai.net/sitemap.xml",
|
| 3 |
"towardsai_page_sitemap_url": "https://towardsai.net/page-sitemap1.xml",
|
|
|
|
| 4 |
"academy_urls": [
|
| 5 |
"https://academy.towardsai.net/",
|
| 6 |
"https://academy.towardsai.net/collections",
|
|
@@ -39,10 +40,42 @@
|
|
| 39 |
"https://towardsai.net/b2b",
|
| 40 |
"https://towardsai.net/let-us-transform-your-team-into-ai-first-employees-to-stay-ahead-of-competitors"
|
| 41 |
],
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 42 |
"extra_external_urls": [
|
| 43 |
"https://www.youtube.com/channel/UCQNjFuhOJM1YqFTPY1Q_kYQ",
|
| 44 |
"https://towardsai.net/towards-ai-resource-library",
|
| 45 |
"https://towardsai.net/book",
|
| 46 |
"https://www.amazon.com/s?k=Towards+AI+Building+LLMs+for+Production"
|
| 47 |
]
|
| 48 |
-
}
|
|
|
|
| 1 |
{
|
| 2 |
"academy_sitemap_url": "https://academy.towardsai.net/sitemap.xml",
|
| 3 |
"towardsai_page_sitemap_url": "https://towardsai.net/page-sitemap1.xml",
|
| 4 |
+
"towardsai_com_page_sitemap_url": "https://towardsai.com/pages-sitemap.xml",
|
| 5 |
"academy_urls": [
|
| 6 |
"https://academy.towardsai.net/",
|
| 7 |
"https://academy.towardsai.net/collections",
|
|
|
|
| 40 |
"https://towardsai.net/b2b",
|
| 41 |
"https://towardsai.net/let-us-transform-your-team-into-ai-first-employees-to-stay-ahead-of-competitors"
|
| 42 |
],
|
| 43 |
+
"reviewed_towardsai_com_pages": [
|
| 44 |
+
"https://towardsai.com/",
|
| 45 |
+
"https://towardsai.com/academy/",
|
| 46 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 47 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 48 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 49 |
+
"https://towardsai.com/academy/python-for-ai-engineering/",
|
| 50 |
+
"https://towardsai.com/academy/ai-for-work/",
|
| 51 |
+
"https://towardsai.com/academy/building-llms-for-production/",
|
| 52 |
+
"https://towardsai.com/academy/bundles/",
|
| 53 |
+
"https://towardsai.com/academy/mentorship/",
|
| 54 |
+
"https://towardsai.com/academy/about/",
|
| 55 |
+
"https://towardsai.com/academy/contact/",
|
| 56 |
+
"https://towardsai.com/contribute/",
|
| 57 |
+
"https://towardsai.com/enterprise/software-developer-to-ai-engineer/",
|
| 58 |
+
"https://towardsai.com/enterprise/agentic-developer-conversion/",
|
| 59 |
+
"https://towardsai.com/academy/affiliate/",
|
| 60 |
+
"https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 61 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 62 |
+
"https://towardsai.com/academy/book/",
|
| 63 |
+
"https://towardsai.com/academy/bundles/get-it-all/",
|
| 64 |
+
"https://towardsai.com/academy/bundles/from-coding-novice-to-advanced-llm-developer/",
|
| 65 |
+
"https://towardsai.com/academy/bundles/10-hour-crash-course-into-llm-developer-expert/",
|
| 66 |
+
"https://towardsai.com/enterpriseenablement/",
|
| 67 |
+
"https://towardsai.com/valuecreation/",
|
| 68 |
+
"https://towardsai.com/valuecreation/careers/",
|
| 69 |
+
"https://towardsai.com/valuecreation/deployment-strategist/",
|
| 70 |
+
"https://towardsai.com/valuecreation/junior-ai-engineer/",
|
| 71 |
+
"https://towardsai.com/valuecreation/senior-ai-engineer/",
|
| 72 |
+
"https://towardsai.com/theaitastegap/",
|
| 73 |
+
"https://towardsai.com/webinars/agentengineering/"
|
| 74 |
+
],
|
| 75 |
"extra_external_urls": [
|
| 76 |
"https://www.youtube.com/channel/UCQNjFuhOJM1YqFTPY1Q_kYQ",
|
| 77 |
"https://towardsai.net/towards-ai-resource-library",
|
| 78 |
"https://towardsai.net/book",
|
| 79 |
"https://www.amazon.com/s?k=Towards+AI+Building+LLMs+for+Production"
|
| 80 |
]
|
| 81 |
+
}
|
data/pages.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/towardsai_com_pages.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
scripts/build_towardsai_com_catalog.py
ADDED
|
@@ -0,0 +1,1342 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import hashlib
|
| 5 |
+
import json
|
| 6 |
+
import re
|
| 7 |
+
from collections.abc import Iterable
|
| 8 |
+
from datetime import UTC, datetime
|
| 9 |
+
from html.parser import HTMLParser
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
from typing import Any
|
| 12 |
+
from urllib.parse import urljoin, urlparse, urlunparse
|
| 13 |
+
from xml.etree import ElementTree
|
| 14 |
+
|
| 15 |
+
import requests
|
| 16 |
+
|
| 17 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 18 |
+
DEFAULT_COM_OUTPUT = ROOT / "data" / "towardsai_com_pages.json"
|
| 19 |
+
DEFAULT_ACADEMY_OUTPUT = ROOT / "data" / "pages.json"
|
| 20 |
+
# Kept for callers of the original, .com-only builder.
|
| 21 |
+
DEFAULT_OUTPUT = DEFAULT_COM_OUTPUT
|
| 22 |
+
|
| 23 |
+
COM_SITEMAP_URL = "https://towardsai.com/pages-sitemap.xml"
|
| 24 |
+
ACADEMY_SITEMAP_URL = "https://academy.towardsai.net/sitemap.xml"
|
| 25 |
+
|
| 26 |
+
COM_HOSTS = frozenset({"towardsai.com", "www.towardsai.com"})
|
| 27 |
+
ACADEMY_HOSTS = frozenset({"academy.towardsai.net"})
|
| 28 |
+
|
| 29 |
+
MIN_CHUNK_CHARS = 1_200
|
| 30 |
+
TARGET_CHUNK_CHARS = 1_550
|
| 31 |
+
MAX_CHUNK_CHARS = 1_800
|
| 32 |
+
MAX_EVIDENCE_SPAN_CHARS = 320
|
| 33 |
+
SENTENCE_SPLIT_RE = re.compile(r"(?<=[.!?])\s+(?=[\"'(\[]*[A-Z0-9])")
|
| 34 |
+
SPAN_TOKEN_RE = re.compile(r"[A-Za-z0-9]+")
|
| 35 |
+
|
| 36 |
+
# Collection pages are still scanned and hashed, but are not factual evidence:
|
| 37 |
+
# their compact cards have contradicted the corresponding canonical detail pages.
|
| 38 |
+
COM_EXCLUSIONS = {
|
| 39 |
+
"/": "homepage offer summaries can conflict with canonical detail pages",
|
| 40 |
+
"/academy": "catalog summary can conflict with canonical offer detail pages",
|
| 41 |
+
"/academy/bundles": (
|
| 42 |
+
"catalog summary can conflict with canonical bundle detail pages"
|
| 43 |
+
),
|
| 44 |
+
}
|
| 45 |
+
|
| 46 |
+
# These Thinkific pages are published in the sitemap, but are unfinished template,
|
| 47 |
+
# collection, or legacy copies of authoritative product pages. They are fetched so
|
| 48 |
+
# that a refresh records their existence and hash, then withheld from retrieval.
|
| 49 |
+
ACADEMY_EXCLUSIONS = {
|
| 50 |
+
"/collections": ("catalog summary can lag behind canonical offer detail pages"),
|
| 51 |
+
"/collections/products": (
|
| 52 |
+
"catalog summary can lag behind canonical offer detail pages"
|
| 53 |
+
),
|
| 54 |
+
"/collections/developers": (
|
| 55 |
+
"catalog summary can lag behind canonical offer detail pages"
|
| 56 |
+
),
|
| 57 |
+
"/collections/professionals": (
|
| 58 |
+
"catalog summary can lag behind canonical offer detail pages"
|
| 59 |
+
),
|
| 60 |
+
"/pages/free-resources": (
|
| 61 |
+
"legacy resource summary conflicts with the current LLM Fundamentals offer"
|
| 62 |
+
),
|
| 63 |
+
"/pages/agent-course-landing-page": (
|
| 64 |
+
"legacy duplicate of the Agent Engineering course page"
|
| 65 |
+
),
|
| 66 |
+
"/pages/agent-course-new-page-cro": (
|
| 67 |
+
"alternate sales copy duplicates the Agent Engineering course page"
|
| 68 |
+
),
|
| 69 |
+
"/pages/towards-ai-insider": "unfinished template content",
|
| 70 |
+
"/pages/webinar": "unfinished template content",
|
| 71 |
+
"/pages/new-home-page": "legacy duplicate of the Academy home page",
|
| 72 |
+
"/pages/landing-page-free-email-course": "unfinished template content",
|
| 73 |
+
"/pages/choose-your-course": "legacy duplicate of the courses collection",
|
| 74 |
+
}
|
| 75 |
+
|
| 76 |
+
# Cross-site identifiers let retrieval prefer the canonical .com description of
|
| 77 |
+
# an offer while retaining the Thinkific purchase/enrolment page as provenance.
|
| 78 |
+
OFFER_PATHS = {
|
| 79 |
+
"/academy/full-stack-ai-engineering": "full-stack-ai-engineering",
|
| 80 |
+
"/courses/beginner-to-advanced-llm-dev": "full-stack-ai-engineering",
|
| 81 |
+
"/academy/agent-engineering": "agent-engineering",
|
| 82 |
+
"/academy/agentic-ai-engineering": "agent-engineering",
|
| 83 |
+
"/courses/agent-engineering": "agent-engineering",
|
| 84 |
+
"/pages/agent-course-landing-page": "agent-engineering",
|
| 85 |
+
"/pages/agent-course-new-page-cro": "agent-engineering",
|
| 86 |
+
"/academy/llm-primer": "llm-primer",
|
| 87 |
+
"/courses/llm-primer": "llm-primer",
|
| 88 |
+
"/academy/python-for-ai-engineering": "python-for-ai-engineering",
|
| 89 |
+
"/courses/python-for-genai": "python-for-ai-engineering",
|
| 90 |
+
"/academy/ai-for-work": "ai-for-work",
|
| 91 |
+
"/courses/ai-business-professionals": "ai-for-work",
|
| 92 |
+
"/academy/building-llms-for-production": "building-llms-for-production",
|
| 93 |
+
"/courses/buildingllmsforproduction": "building-llms-for-production",
|
| 94 |
+
"/academy/mentorship": "mentorship",
|
| 95 |
+
"/bundles/tai-mentorship": "mentorship",
|
| 96 |
+
"/academy/bundles/get-it-all": "get-it-all",
|
| 97 |
+
"/bundles/get-it-all": "get-it-all",
|
| 98 |
+
"/academy/bundles/from-coding-novice-to-advanced-llm-developer": (
|
| 99 |
+
"from-coding-novice-to-advanced-llm-developer"
|
| 100 |
+
),
|
| 101 |
+
"/bundles/from-coding-novice-to-advanced-llm-developer": (
|
| 102 |
+
"from-coding-novice-to-advanced-llm-developer"
|
| 103 |
+
),
|
| 104 |
+
"/academy/bundles/10-hour-crash-course-into-llm-developer-expert": (
|
| 105 |
+
"10-hour-crash-course-into-llm-developer-expert"
|
| 106 |
+
),
|
| 107 |
+
"/bundles/10-hour-crash-course-into-llm-developer-expert": (
|
| 108 |
+
"10-hour-crash-course-into-llm-developer-expert"
|
| 109 |
+
),
|
| 110 |
+
"/academy/agent-engineering-free-preview": ("agent-engineering-free-preview"),
|
| 111 |
+
"/academy/full-stack-ai-engineering-free-preview": (
|
| 112 |
+
"full-stack-ai-engineering-free-preview"
|
| 113 |
+
),
|
| 114 |
+
"/products/digital_downloads/agents-cheatsheet": "agents-cheatsheet",
|
| 115 |
+
"/products/digital_downloads/anti-slop-framework": "anti-slop-framework",
|
| 116 |
+
}
|
| 117 |
+
|
| 118 |
+
# These two resources do not belong to either first-party sitemap. Keeping the
|
| 119 |
+
# small, explicit allowlist avoids turning arbitrary outbound links into evidence.
|
| 120 |
+
MANUAL_SAFE_RESOURCES: tuple[dict[str, Any], ...] = (
|
| 121 |
+
{
|
| 122 |
+
"url": ("https://www.amazon.com/s?k=Towards+AI+Building+LLMs+for+Production"),
|
| 123 |
+
"kind": "book_external",
|
| 124 |
+
"title": "Towards AI book on Amazon",
|
| 125 |
+
"text": (
|
| 126 |
+
"Building LLMs for Production by Towards AI — Amazon search resource."
|
| 127 |
+
),
|
| 128 |
+
"links": [
|
| 129 |
+
{
|
| 130 |
+
"text": "Amazon book search",
|
| 131 |
+
"url": (
|
| 132 |
+
"https://www.amazon.com/s?k=Towards+AI+Building+LLMs+for+Production"
|
| 133 |
+
),
|
| 134 |
+
}
|
| 135 |
+
],
|
| 136 |
+
},
|
| 137 |
+
{
|
| 138 |
+
"url": "https://www.youtube.com/channel/UCQNjFuhOJM1YqFTPY1Q_kYQ",
|
| 139 |
+
"kind": "free_content_external",
|
| 140 |
+
"title": "What's AI YouTube channel",
|
| 141 |
+
"text": ("What's AI by Louis-François Bouchard — YouTube channel."),
|
| 142 |
+
"links": [
|
| 143 |
+
{
|
| 144 |
+
"text": "What's AI on YouTube",
|
| 145 |
+
"url": ("https://www.youtube.com/channel/UCQNjFuhOJM1YqFTPY1Q_kYQ"),
|
| 146 |
+
}
|
| 147 |
+
],
|
| 148 |
+
},
|
| 149 |
+
)
|
| 150 |
+
|
| 151 |
+
# Human-reviewed routing metadata is preserved across sitemap refreshes, but the
|
| 152 |
+
# summary is metadata only: factual answers must cite scraped page chunks.
|
| 153 |
+
CURATED_COM_METADATA: tuple[dict[str, str], ...] = (
|
| 154 |
+
{
|
| 155 |
+
"url": "https://towardsai.com/",
|
| 156 |
+
"review_title": "Homepage",
|
| 157 |
+
"kind": "page",
|
| 158 |
+
"reviewed_summary": "Towards AI deployment, education, and company overview.",
|
| 159 |
+
},
|
| 160 |
+
{
|
| 161 |
+
"url": "https://towardsai.com/academy/",
|
| 162 |
+
"review_title": "Academy hub",
|
| 163 |
+
"kind": "collection",
|
| 164 |
+
"reviewed_summary": "Central comparison page for Towards AI learning paths.",
|
| 165 |
+
},
|
| 166 |
+
{
|
| 167 |
+
"url": "https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 168 |
+
"review_title": "Full Stack AI Engineering",
|
| 169 |
+
"kind": "course",
|
| 170 |
+
"reviewed_summary": "Production LLM product engineering course.",
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"url": "https://towardsai.com/academy/agent-engineering/",
|
| 174 |
+
"review_title": "Agent Engineering",
|
| 175 |
+
"kind": "course",
|
| 176 |
+
"reviewed_summary": "Production-focused agent engineering course.",
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"url": "https://towardsai.com/academy/llm-primer/",
|
| 180 |
+
"review_title": "10-Hour LLM Fundamentals",
|
| 181 |
+
"kind": "course",
|
| 182 |
+
"reviewed_summary": "Focused LLM fundamentals video course.",
|
| 183 |
+
},
|
| 184 |
+
{
|
| 185 |
+
"url": "https://towardsai.com/academy/python-for-ai-engineering/",
|
| 186 |
+
"review_title": "Python for AI Engineering",
|
| 187 |
+
"kind": "course",
|
| 188 |
+
"reviewed_summary": "Beginner Python foundations for AI engineering.",
|
| 189 |
+
},
|
| 190 |
+
{
|
| 191 |
+
"url": "https://towardsai.com/academy/ai-for-work/",
|
| 192 |
+
"review_title": "AI for Work",
|
| 193 |
+
"kind": "course",
|
| 194 |
+
"reviewed_summary": "Practical no-code AI training for professionals.",
|
| 195 |
+
},
|
| 196 |
+
{
|
| 197 |
+
"url": "https://towardsai.com/academy/building-llms-for-production/",
|
| 198 |
+
"review_title": "Building LLMs for Production",
|
| 199 |
+
"kind": "book",
|
| 200 |
+
"reviewed_summary": "Towards AI book and companion learning page.",
|
| 201 |
+
},
|
| 202 |
+
{
|
| 203 |
+
"url": "https://towardsai.com/academy/book/",
|
| 204 |
+
"review_title": "The Book",
|
| 205 |
+
"kind": "book",
|
| 206 |
+
"reviewed_summary": "Towards AI book page.",
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"url": "https://towardsai.com/academy/bundles/",
|
| 210 |
+
"review_title": "Academy bundles",
|
| 211 |
+
"kind": "collection",
|
| 212 |
+
"reviewed_summary": "Central collection of current Academy bundles.",
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 216 |
+
"review_title": "Mentorship",
|
| 217 |
+
"kind": "mentorship",
|
| 218 |
+
"reviewed_summary": (
|
| 219 |
+
"Mentorship includes one course, the 10-Hour LLM Fundamentals video "
|
| 220 |
+
"course from day one; the other currently listed courses are 25% off."
|
| 221 |
+
),
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"url": "https://towardsai.com/academy/about/",
|
| 225 |
+
"review_title": "About",
|
| 226 |
+
"kind": "page",
|
| 227 |
+
"reviewed_summary": "Towards AI Academy and team background.",
|
| 228 |
+
},
|
| 229 |
+
{
|
| 230 |
+
"url": "https://towardsai.com/academy/contact/",
|
| 231 |
+
"review_title": "Contact",
|
| 232 |
+
"kind": "page",
|
| 233 |
+
"reviewed_summary": "Towards AI Academy contact options.",
|
| 234 |
+
},
|
| 235 |
+
{
|
| 236 |
+
"url": "https://towardsai.com/academy/affiliate/",
|
| 237 |
+
"review_title": "Affiliate",
|
| 238 |
+
"kind": "page",
|
| 239 |
+
"reviewed_summary": "Towards AI referral and affiliate program.",
|
| 240 |
+
},
|
| 241 |
+
{
|
| 242 |
+
"url": "https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 243 |
+
"review_title": "Full Stack AI Engineering free preview",
|
| 244 |
+
"kind": "free_resource",
|
| 245 |
+
"reviewed_summary": "Free preview of Full Stack AI Engineering.",
|
| 246 |
+
},
|
| 247 |
+
{
|
| 248 |
+
"url": "https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 249 |
+
"review_title": "Agent Engineering free preview",
|
| 250 |
+
"kind": "free_resource",
|
| 251 |
+
"reviewed_summary": "Free preview of Agent Engineering.",
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"url": "https://towardsai.com/academy/bundles/get-it-all/",
|
| 255 |
+
"review_title": "Get It All",
|
| 256 |
+
"kind": "bundle",
|
| 257 |
+
"reviewed_summary": "Broadest Academy course bundle.",
|
| 258 |
+
},
|
| 259 |
+
{
|
| 260 |
+
"url": "https://towardsai.com/academy/bundles/from-coding-novice-to-advanced-llm-developer/",
|
| 261 |
+
"review_title": "From Non-Coder to AI Engineer",
|
| 262 |
+
"kind": "bundle",
|
| 263 |
+
"reviewed_summary": "Bundle path from Python foundations to AI engineering.",
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"url": "https://towardsai.com/academy/bundles/10-hour-crash-course-into-llm-developer-expert/",
|
| 267 |
+
"review_title": "From Developer to Advanced AI Engineer",
|
| 268 |
+
"kind": "bundle",
|
| 269 |
+
"reviewed_summary": "Bundle path from LLM fundamentals to advanced engineering.",
|
| 270 |
+
},
|
| 271 |
+
{
|
| 272 |
+
"url": "https://towardsai.com/enterprise/software-developer-to-ai-engineer/",
|
| 273 |
+
"review_title": "Software Developer to AI Engineer",
|
| 274 |
+
"kind": "b2b",
|
| 275 |
+
"reviewed_summary": "Enterprise AI-engineer conversion program.",
|
| 276 |
+
},
|
| 277 |
+
{
|
| 278 |
+
"url": "https://towardsai.com/enterprise/agentic-developer-conversion/",
|
| 279 |
+
"review_title": "Agentic Developer Conversion",
|
| 280 |
+
"kind": "b2b",
|
| 281 |
+
"reviewed_summary": "Enterprise coding-agent adoption program.",
|
| 282 |
+
},
|
| 283 |
+
{
|
| 284 |
+
"url": "https://towardsai.com/enterpriseenablement/",
|
| 285 |
+
"review_title": "Enterprise Enablement",
|
| 286 |
+
"kind": "b2b",
|
| 287 |
+
"reviewed_summary": "Enterprise AI training and enablement.",
|
| 288 |
+
},
|
| 289 |
+
{
|
| 290 |
+
"url": "https://towardsai.com/valuecreation/",
|
| 291 |
+
"review_title": "Value Creation",
|
| 292 |
+
"kind": "b2b",
|
| 293 |
+
"reviewed_summary": "Custom AI development and value-creation consulting.",
|
| 294 |
+
},
|
| 295 |
+
{
|
| 296 |
+
"url": "https://towardsai.com/webinars/agentengineering/",
|
| 297 |
+
"review_title": "Agent Engineering webinar",
|
| 298 |
+
"kind": "free_resource",
|
| 299 |
+
"reviewed_summary": "Free agent engineering webinar.",
|
| 300 |
+
},
|
| 301 |
+
)
|
| 302 |
+
|
| 303 |
+
SUPPRESSED_TAGS = frozenset(
|
| 304 |
+
{"script", "style", "noscript", "svg", "template", "nav", "footer"}
|
| 305 |
+
)
|
| 306 |
+
VOID_TAGS = frozenset(
|
| 307 |
+
{
|
| 308 |
+
"area",
|
| 309 |
+
"base",
|
| 310 |
+
"br",
|
| 311 |
+
"col",
|
| 312 |
+
"embed",
|
| 313 |
+
"hr",
|
| 314 |
+
"img",
|
| 315 |
+
"input",
|
| 316 |
+
"link",
|
| 317 |
+
"meta",
|
| 318 |
+
"param",
|
| 319 |
+
"source",
|
| 320 |
+
"track",
|
| 321 |
+
"wbr",
|
| 322 |
+
}
|
| 323 |
+
)
|
| 324 |
+
BLOCK_TAGS = frozenset(
|
| 325 |
+
{
|
| 326 |
+
"address",
|
| 327 |
+
"article",
|
| 328 |
+
"blockquote",
|
| 329 |
+
"dd",
|
| 330 |
+
"div",
|
| 331 |
+
"dl",
|
| 332 |
+
"dt",
|
| 333 |
+
"figcaption",
|
| 334 |
+
"figure",
|
| 335 |
+
"li",
|
| 336 |
+
"main",
|
| 337 |
+
"p",
|
| 338 |
+
"section",
|
| 339 |
+
"td",
|
| 340 |
+
"th",
|
| 341 |
+
}
|
| 342 |
+
)
|
| 343 |
+
HEADING_TAGS = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"})
|
| 344 |
+
SUPPRESSED_ROLES = frozenset({"navigation", "contentinfo"})
|
| 345 |
+
SUPPRESSED_CLASS_TOKENS = frozenset(
|
| 346 |
+
{
|
| 347 |
+
"cookie-banner",
|
| 348 |
+
"cookie-consent",
|
| 349 |
+
"footer",
|
| 350 |
+
"global-footer",
|
| 351 |
+
"global-nav",
|
| 352 |
+
"header",
|
| 353 |
+
"navbar",
|
| 354 |
+
"navigation",
|
| 355 |
+
"site-footer",
|
| 356 |
+
"site-nav",
|
| 357 |
+
"ta-announce",
|
| 358 |
+
# Testimonials, endorsements, simulated conversations, and competitor
|
| 359 |
+
# comparison tables are useful marketing context but are not first-party
|
| 360 |
+
# offer terms. Keeping them citable lets review copy or competitor facts
|
| 361 |
+
# conflict with the actual product specification.
|
| 362 |
+
"ta-ment-thread",
|
| 363 |
+
"ta-ment-vsnote",
|
| 364 |
+
"ta-ment-vswrap",
|
| 365 |
+
"ta-mobile-menu",
|
| 366 |
+
"ta-quote",
|
| 367 |
+
"ta-review",
|
| 368 |
+
}
|
| 369 |
+
)
|
| 370 |
+
|
| 371 |
+
PREVIEW_SUPPRESSED_SECTION_IDS: dict[str, frozenset[str]] = {
|
| 372 |
+
# These sections explicitly describe the paid course, not the free preview.
|
| 373 |
+
"/academy/agent-engineering-free-preview": frozenset({"fullcourse"}),
|
| 374 |
+
"/academy/full-stack-ai-engineering-free-preview": frozenset(
|
| 375 |
+
{"outcomes", "cta"}
|
| 376 |
+
),
|
| 377 |
+
}
|
| 378 |
+
PAGE_SUPPRESSED_CLASS_TOKENS: dict[str, frozenset[str]] = {
|
| 379 |
+
# Keep the first-party feature list in the comparison section, but discard
|
| 380 |
+
# competitor scorecards and bought-separately summaries.
|
| 381 |
+
"/academy/mentorship": frozenset({"board", "sum"}),
|
| 382 |
+
}
|
| 383 |
+
|
| 384 |
+
|
| 385 |
+
def _clean(text: str) -> str:
|
| 386 |
+
return re.sub(r"\s+", " ", text).strip()
|
| 387 |
+
|
| 388 |
+
|
| 389 |
+
def _utc_now() -> str:
|
| 390 |
+
return datetime.now(UTC).isoformat()
|
| 391 |
+
|
| 392 |
+
|
| 393 |
+
def _local_name(tag: str) -> str:
|
| 394 |
+
return tag.rsplit("}", 1)[-1].lower()
|
| 395 |
+
|
| 396 |
+
|
| 397 |
+
def _normalise_url(url: str) -> str:
|
| 398 |
+
"""Return a fragment-free URL with a normalised scheme and hostname."""
|
| 399 |
+
|
| 400 |
+
parsed = urlparse(url.strip())
|
| 401 |
+
scheme = parsed.scheme.lower()
|
| 402 |
+
hostname = (parsed.hostname or "").lower()
|
| 403 |
+
if not scheme or not hostname:
|
| 404 |
+
return url.strip()
|
| 405 |
+
port = parsed.port
|
| 406 |
+
if port and not (
|
| 407 |
+
(scheme == "http" and port == 80) or (scheme == "https" and port == 443)
|
| 408 |
+
):
|
| 409 |
+
netloc = f"{hostname}:{port}"
|
| 410 |
+
else:
|
| 411 |
+
netloc = hostname
|
| 412 |
+
path = parsed.path or "/"
|
| 413 |
+
return urlunparse((scheme, netloc, path, "", parsed.query, ""))
|
| 414 |
+
|
| 415 |
+
|
| 416 |
+
def _path(url: str) -> str:
|
| 417 |
+
return urlparse(url).path.rstrip("/") or "/"
|
| 418 |
+
|
| 419 |
+
|
| 420 |
+
def _sha256(text: str) -> str:
|
| 421 |
+
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
| 422 |
+
|
| 423 |
+
|
| 424 |
+
def _evidence_hash(page: dict[str, Any]) -> str:
|
| 425 |
+
canonical = json.dumps(
|
| 426 |
+
{key: value for key, value in page.items() if key != "evidence_hash"},
|
| 427 |
+
ensure_ascii=False,
|
| 428 |
+
sort_keys=True,
|
| 429 |
+
separators=(",", ":"),
|
| 430 |
+
)
|
| 431 |
+
return _sha256(canonical)
|
| 432 |
+
|
| 433 |
+
|
| 434 |
+
def _is_suppressed_container(tag: str, attributes: dict[str, str | None]) -> bool:
|
| 435 |
+
class_tokens = {
|
| 436 |
+
token
|
| 437 |
+
for token in re.split(r"[^a-z0-9_-]+", (attributes.get("class") or "").lower())
|
| 438 |
+
if token
|
| 439 |
+
}
|
| 440 |
+
# Offer pages use a semantic <header class="ta-hero"> for product stats,
|
| 441 |
+
# pricing, and eligibility facts. Preserve that content header while still
|
| 442 |
+
# suppressing the site's navigation header.
|
| 443 |
+
content_header = tag == "header" and any(
|
| 444 |
+
token == "hero" or token.endswith("-hero") for token in class_tokens
|
| 445 |
+
)
|
| 446 |
+
if tag in SUPPRESSED_TAGS and not content_header:
|
| 447 |
+
return True
|
| 448 |
+
if (attributes.get("role") or "").lower() in SUPPRESSED_ROLES:
|
| 449 |
+
return True
|
| 450 |
+
tokens = re.split(
|
| 451 |
+
r"[^a-z0-9_-]+",
|
| 452 |
+
" ".join([attributes.get("id") or "", attributes.get("class") or ""]).lower(),
|
| 453 |
+
)
|
| 454 |
+
return any(token in SUPPRESSED_CLASS_TOKENS for token in tokens if token)
|
| 455 |
+
|
| 456 |
+
|
| 457 |
+
def _is_page_specific_suppressed_container(
|
| 458 |
+
base_url: str, tag: str, attributes: dict[str, str | None]
|
| 459 |
+
) -> bool:
|
| 460 |
+
page_path = _path(base_url)
|
| 461 |
+
if tag == "section":
|
| 462 |
+
suppressed_ids = PREVIEW_SUPPRESSED_SECTION_IDS.get(page_path, frozenset())
|
| 463 |
+
if (attributes.get("id") or "").casefold() in suppressed_ids:
|
| 464 |
+
return True
|
| 465 |
+
class_tokens = {
|
| 466 |
+
token
|
| 467 |
+
for token in re.split(
|
| 468 |
+
r"[^a-z0-9_-]+", (attributes.get("class") or "").casefold()
|
| 469 |
+
)
|
| 470 |
+
if token
|
| 471 |
+
}
|
| 472 |
+
return bool(
|
| 473 |
+
class_tokens & PAGE_SUPPRESSED_CLASS_TOKENS.get(page_path, frozenset())
|
| 474 |
+
)
|
| 475 |
+
|
| 476 |
+
|
| 477 |
+
def _meta_refresh_target(content: str) -> str:
|
| 478 |
+
for part in content.split(";"):
|
| 479 |
+
key, separator, value = part.strip().partition("=")
|
| 480 |
+
if separator and key.strip().lower() == "url":
|
| 481 |
+
return value.strip().strip("\"'")
|
| 482 |
+
return ""
|
| 483 |
+
|
| 484 |
+
|
| 485 |
+
class PageParser(HTMLParser):
|
| 486 |
+
"""Extract visible, sectioned page content and canonicalisation hints."""
|
| 487 |
+
|
| 488 |
+
def __init__(self, base_url: str) -> None:
|
| 489 |
+
super().__init__(convert_charrefs=True)
|
| 490 |
+
self.base_url = base_url
|
| 491 |
+
self.suppressed_depth = 0
|
| 492 |
+
self.head_depth = 0
|
| 493 |
+
self.title_parts: list[str] = []
|
| 494 |
+
self.text_parts: list[str] = []
|
| 495 |
+
self.headings: list[str] = []
|
| 496 |
+
self.links: list[dict[str, str]] = []
|
| 497 |
+
self.sections: list[dict[str, Any]] = []
|
| 498 |
+
self.meta_description = ""
|
| 499 |
+
self.canonical_url = ""
|
| 500 |
+
self.meta_refresh_url = ""
|
| 501 |
+
self._in_title = False
|
| 502 |
+
self._heading_parts: list[str] | None = None
|
| 503 |
+
self._current_heading = ""
|
| 504 |
+
self._section_parts: list[str] = []
|
| 505 |
+
self._section_spans: list[str] = []
|
| 506 |
+
self._block_parts: list[str] = []
|
| 507 |
+
self._table_row_parts: list[str] | None = None
|
| 508 |
+
self._table_head_depth = 0
|
| 509 |
+
self._link_href = ""
|
| 510 |
+
self._link_parts: list[str] | None = None
|
| 511 |
+
|
| 512 |
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
| 513 |
+
tag = tag.lower()
|
| 514 |
+
values = {key.lower(): value for key, value in attrs}
|
| 515 |
+
|
| 516 |
+
if self.suppressed_depth:
|
| 517 |
+
if tag not in VOID_TAGS:
|
| 518 |
+
self.suppressed_depth += 1
|
| 519 |
+
return
|
| 520 |
+
if _is_suppressed_container(tag, values) or _is_page_specific_suppressed_container(
|
| 521 |
+
self.base_url, tag, values
|
| 522 |
+
):
|
| 523 |
+
self._flush_block()
|
| 524 |
+
self.suppressed_depth = 1
|
| 525 |
+
return
|
| 526 |
+
|
| 527 |
+
if tag == "head":
|
| 528 |
+
self.head_depth += 1
|
| 529 |
+
if tag == "title":
|
| 530 |
+
self._in_title = True
|
| 531 |
+
elif tag == "link":
|
| 532 |
+
rel = {item.lower() for item in (values.get("rel") or "").split()}
|
| 533 |
+
if "canonical" in rel and not self.canonical_url:
|
| 534 |
+
self.canonical_url = (values.get("href") or "").strip()
|
| 535 |
+
elif tag == "meta":
|
| 536 |
+
name = (values.get("name") or "").lower()
|
| 537 |
+
http_equiv = (values.get("http-equiv") or "").lower()
|
| 538 |
+
content = values.get("content") or ""
|
| 539 |
+
if name == "description":
|
| 540 |
+
self.meta_description = _clean(content)
|
| 541 |
+
elif http_equiv == "refresh" and not self.meta_refresh_url:
|
| 542 |
+
self.meta_refresh_url = _meta_refresh_target(content)
|
| 543 |
+
|
| 544 |
+
if self.head_depth:
|
| 545 |
+
return
|
| 546 |
+
if tag == "thead":
|
| 547 |
+
self._table_head_depth += 1
|
| 548 |
+
elif tag == "tr":
|
| 549 |
+
self._flush_block()
|
| 550 |
+
self._table_row_parts = []
|
| 551 |
+
elif tag in HEADING_TAGS:
|
| 552 |
+
self._flush_block()
|
| 553 |
+
self._flush_section()
|
| 554 |
+
self._heading_parts = []
|
| 555 |
+
elif tag == "a":
|
| 556 |
+
self._link_href = values.get("href") or ""
|
| 557 |
+
self._link_parts = []
|
| 558 |
+
elif tag == "br":
|
| 559 |
+
self._flush_block()
|
| 560 |
+
|
| 561 |
+
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
| 562 |
+
# Avoid changing suppression depth for an explicitly self-closing tag.
|
| 563 |
+
if self.suppressed_depth:
|
| 564 |
+
return
|
| 565 |
+
self.handle_starttag(tag, attrs)
|
| 566 |
+
if tag.lower() not in VOID_TAGS:
|
| 567 |
+
self.handle_endtag(tag)
|
| 568 |
+
|
| 569 |
+
def handle_endtag(self, tag: str) -> None:
|
| 570 |
+
tag = tag.lower()
|
| 571 |
+
if self.suppressed_depth:
|
| 572 |
+
self.suppressed_depth -= 1
|
| 573 |
+
return
|
| 574 |
+
if tag == "head":
|
| 575 |
+
self.head_depth = max(0, self.head_depth - 1)
|
| 576 |
+
return
|
| 577 |
+
if tag == "title":
|
| 578 |
+
self._in_title = False
|
| 579 |
+
return
|
| 580 |
+
if self.head_depth:
|
| 581 |
+
return
|
| 582 |
+
if tag in HEADING_TAGS and self._heading_parts is not None:
|
| 583 |
+
heading = _clean(" ".join(self._heading_parts))
|
| 584 |
+
if heading:
|
| 585 |
+
self.headings.append(heading)
|
| 586 |
+
self._current_heading = heading
|
| 587 |
+
self._heading_parts = None
|
| 588 |
+
elif tag == "a" and self._link_parts is not None:
|
| 589 |
+
href = self._link_href.strip()
|
| 590 |
+
if href and not href.lower().startswith(
|
| 591 |
+
("#", "mailto:", "tel:", "javascript:")
|
| 592 |
+
):
|
| 593 |
+
self.links.append(
|
| 594 |
+
{
|
| 595 |
+
"text": _clean(" ".join(self._link_parts)),
|
| 596 |
+
"url": _normalise_url(urljoin(self.base_url, href)),
|
| 597 |
+
}
|
| 598 |
+
)
|
| 599 |
+
self._link_href = ""
|
| 600 |
+
self._link_parts = None
|
| 601 |
+
if tag in BLOCK_TAGS:
|
| 602 |
+
self._flush_block()
|
| 603 |
+
if tag == "tr" and self._table_row_parts is not None:
|
| 604 |
+
self._flush_block()
|
| 605 |
+
row = _clean(" ".join(self._table_row_parts))
|
| 606 |
+
if row and self._table_head_depth == 0:
|
| 607 |
+
self._section_spans.append(row)
|
| 608 |
+
self._table_row_parts = None
|
| 609 |
+
if tag == "thead":
|
| 610 |
+
self._table_head_depth = max(0, self._table_head_depth - 1)
|
| 611 |
+
|
| 612 |
+
def handle_data(self, data: str) -> None:
|
| 613 |
+
value = _clean(data)
|
| 614 |
+
if not value or self.suppressed_depth:
|
| 615 |
+
return
|
| 616 |
+
if self._in_title:
|
| 617 |
+
self.title_parts.append(value)
|
| 618 |
+
return
|
| 619 |
+
if self.head_depth:
|
| 620 |
+
return
|
| 621 |
+
if self._heading_parts is not None:
|
| 622 |
+
self._heading_parts.append(value)
|
| 623 |
+
return
|
| 624 |
+
self.text_parts.append(value)
|
| 625 |
+
self._block_parts.append(value)
|
| 626 |
+
if self._table_row_parts is not None:
|
| 627 |
+
self._table_row_parts.append(value)
|
| 628 |
+
if self._link_parts is not None:
|
| 629 |
+
self._link_parts.append(value)
|
| 630 |
+
|
| 631 |
+
def close(self) -> None:
|
| 632 |
+
super().close()
|
| 633 |
+
self._flush_block()
|
| 634 |
+
self._flush_section()
|
| 635 |
+
|
| 636 |
+
def _flush_block(self) -> None:
|
| 637 |
+
block = _clean(" ".join(self._block_parts))
|
| 638 |
+
if block:
|
| 639 |
+
self._section_parts.append(block)
|
| 640 |
+
if self._table_row_parts is None:
|
| 641 |
+
self._section_spans.append(block)
|
| 642 |
+
self._block_parts = []
|
| 643 |
+
|
| 644 |
+
def _flush_section(self) -> None:
|
| 645 |
+
self._flush_block()
|
| 646 |
+
body = _clean(" ".join(self._section_parts))
|
| 647 |
+
if body or self._current_heading:
|
| 648 |
+
self.sections.append(
|
| 649 |
+
{
|
| 650 |
+
"heading": self._current_heading,
|
| 651 |
+
"text": body,
|
| 652 |
+
"spans": list(dict.fromkeys(self._section_spans)),
|
| 653 |
+
}
|
| 654 |
+
)
|
| 655 |
+
self._section_parts = []
|
| 656 |
+
self._section_spans = []
|
| 657 |
+
|
| 658 |
+
|
| 659 |
+
def _unique_links(links: Iterable[dict[str, str]]) -> list[dict[str, str]]:
|
| 660 |
+
result: list[dict[str, str]] = []
|
| 661 |
+
seen: set[tuple[str, str]] = set()
|
| 662 |
+
for link in links:
|
| 663 |
+
key = (link["text"], link["url"])
|
| 664 |
+
if key not in seen:
|
| 665 |
+
seen.add(key)
|
| 666 |
+
result.append(link)
|
| 667 |
+
return result
|
| 668 |
+
|
| 669 |
+
|
| 670 |
+
def _split_to_limit(text: str, limit: int) -> list[str]:
|
| 671 |
+
"""Split text without breaking words, preferring paragraph/sentence edges."""
|
| 672 |
+
|
| 673 |
+
text = text.strip()
|
| 674 |
+
pieces: list[str] = []
|
| 675 |
+
while len(text) > limit:
|
| 676 |
+
lower_bound = max(1, int(limit * 0.62))
|
| 677 |
+
candidates = [
|
| 678 |
+
text.rfind("\n\n", lower_bound, limit + 1),
|
| 679 |
+
text.rfind(". ", lower_bound, limit + 1),
|
| 680 |
+
text.rfind("? ", lower_bound, limit + 1),
|
| 681 |
+
text.rfind("! ", lower_bound, limit + 1),
|
| 682 |
+
text.rfind(" ", lower_bound, limit + 1),
|
| 683 |
+
]
|
| 684 |
+
cut = max(candidates)
|
| 685 |
+
if cut < 1:
|
| 686 |
+
cut = text.rfind(" ", 0, limit + 1)
|
| 687 |
+
if cut < 1:
|
| 688 |
+
cut = limit
|
| 689 |
+
elif text[cut : cut + 2] in {". ", "? ", "! "}:
|
| 690 |
+
cut += 1
|
| 691 |
+
pieces.append(text[:cut].strip())
|
| 692 |
+
text = text[cut:].strip()
|
| 693 |
+
if text:
|
| 694 |
+
pieces.append(text)
|
| 695 |
+
return pieces
|
| 696 |
+
|
| 697 |
+
|
| 698 |
+
def _atomic_span_texts(raw_spans: Iterable[str]) -> list[str]:
|
| 699 |
+
"""Return complete, bounded source spans; never emit arbitrary substrings."""
|
| 700 |
+
|
| 701 |
+
result: list[str] = []
|
| 702 |
+
seen: set[str] = set()
|
| 703 |
+
for raw_span in raw_spans:
|
| 704 |
+
block = _clean(str(raw_span))
|
| 705 |
+
if not block:
|
| 706 |
+
continue
|
| 707 |
+
# Accordion rows are flattened by the parser as ``Question? + Answer``.
|
| 708 |
+
# Only the complete answer is evidence: the question may contain a false
|
| 709 |
+
# premise, comparison price, or visitor-style wording that must never be
|
| 710 |
+
# treated as a first-party offer fact.
|
| 711 |
+
faq_parts = re.split(r"(?<=\?)\s*\+\s*", block, maxsplit=1)
|
| 712 |
+
if len(faq_parts) > 1:
|
| 713 |
+
candidates = faq_parts[1:]
|
| 714 |
+
else:
|
| 715 |
+
sentences = SENTENCE_SPLIT_RE.split(block)
|
| 716 |
+
candidates = sentences if len(sentences) > 1 else [block]
|
| 717 |
+
for candidate in candidates:
|
| 718 |
+
span = _clean(candidate)
|
| 719 |
+
key = span.casefold()
|
| 720 |
+
if (
|
| 721 |
+
len(SPAN_TOKEN_RE.findall(span)) < 2
|
| 722 |
+
or len(span) > MAX_EVIDENCE_SPAN_CHARS
|
| 723 |
+
or any(symbol in span for symbol in ("✓", "✗"))
|
| 724 |
+
or key in seen
|
| 725 |
+
):
|
| 726 |
+
continue
|
| 727 |
+
seen.add(key)
|
| 728 |
+
result.append(span)
|
| 729 |
+
return result
|
| 730 |
+
|
| 731 |
+
|
| 732 |
+
def _span_records(
|
| 733 |
+
canonical_url: str, chunk_id: str, raw_spans: Iterable[str]
|
| 734 |
+
) -> list[dict[str, str]]:
|
| 735 |
+
records: list[dict[str, str]] = []
|
| 736 |
+
for text in _atomic_span_texts(raw_spans):
|
| 737 |
+
identity = f"{canonical_url}\n{chunk_id}\n{text}"
|
| 738 |
+
records.append(
|
| 739 |
+
{
|
| 740 |
+
"span_id": f"span-{_sha256(identity)[:24]}",
|
| 741 |
+
"text": text,
|
| 742 |
+
}
|
| 743 |
+
)
|
| 744 |
+
return records
|
| 745 |
+
|
| 746 |
+
|
| 747 |
+
def build_chunks(
|
| 748 |
+
sections: Iterable[dict[str, Any]],
|
| 749 |
+
canonical_url: str,
|
| 750 |
+
*,
|
| 751 |
+
min_chars: int = MIN_CHUNK_CHARS,
|
| 752 |
+
target_chars: int = TARGET_CHUNK_CHARS,
|
| 753 |
+
max_chars: int = MAX_CHUNK_CHARS,
|
| 754 |
+
) -> list[dict[str, Any]]:
|
| 755 |
+
"""Build bounded deterministic chunks while retaining section headings."""
|
| 756 |
+
|
| 757 |
+
if not 0 < min_chars <= target_chars <= max_chars:
|
| 758 |
+
raise ValueError("chunk sizes must satisfy 0 < min <= target <= max")
|
| 759 |
+
|
| 760 |
+
pieces: list[dict[str, Any]] = []
|
| 761 |
+
for section_index, section in enumerate(sections):
|
| 762 |
+
heading = _clean(str(section.get("heading", "")))
|
| 763 |
+
body = _clean(str(section.get("text", "")))
|
| 764 |
+
raw_section_spans = [str(item) for item in section.get("spans", [])]
|
| 765 |
+
section_spans = _atomic_span_texts(raw_section_spans)
|
| 766 |
+
heading_spans = _atomic_span_texts([heading]) if heading else []
|
| 767 |
+
if not heading and not body:
|
| 768 |
+
continue
|
| 769 |
+
prefix = f"{heading}\n" if heading else ""
|
| 770 |
+
body_limit = max(1, max_chars - len(prefix))
|
| 771 |
+
body_pieces = _split_to_limit(body, body_limit) if body else [""]
|
| 772 |
+
for part_index, body_piece in enumerate(body_pieces):
|
| 773 |
+
text = f"{prefix}{body_piece}".strip()
|
| 774 |
+
normalized_piece = _clean(body_piece)
|
| 775 |
+
piece_spans = [span for span in section_spans if span in normalized_piece]
|
| 776 |
+
if part_index == 0:
|
| 777 |
+
piece_spans = [
|
| 778 |
+
*[span for span in heading_spans if re.search(r"\d|[$€£¥%]", span)],
|
| 779 |
+
*piece_spans,
|
| 780 |
+
]
|
| 781 |
+
if not piece_spans:
|
| 782 |
+
piece_spans.extend(heading_spans)
|
| 783 |
+
pieces.append(
|
| 784 |
+
{
|
| 785 |
+
"text": text,
|
| 786 |
+
"heading": heading or "Overview",
|
| 787 |
+
"key": f"section-{section_index}-part-{part_index}",
|
| 788 |
+
"spans": piece_spans,
|
| 789 |
+
}
|
| 790 |
+
)
|
| 791 |
+
|
| 792 |
+
groups: list[list[dict[str, Any]]] = []
|
| 793 |
+
current: list[dict[str, Any]] = []
|
| 794 |
+
current_length = 0
|
| 795 |
+
for piece in pieces:
|
| 796 |
+
separator_length = 2 if current else 0
|
| 797 |
+
candidate_length = current_length + separator_length + len(piece["text"])
|
| 798 |
+
if current and candidate_length > max_chars:
|
| 799 |
+
groups.append(current)
|
| 800 |
+
current = []
|
| 801 |
+
current_length = 0
|
| 802 |
+
separator_length = 0
|
| 803 |
+
current.append(piece)
|
| 804 |
+
current_length += separator_length + len(piece["text"])
|
| 805 |
+
if current_length >= target_chars:
|
| 806 |
+
groups.append(current)
|
| 807 |
+
current = []
|
| 808 |
+
current_length = 0
|
| 809 |
+
if current:
|
| 810 |
+
groups.append(current)
|
| 811 |
+
|
| 812 |
+
# Fold a small final group into its predecessor when the hard maximum allows.
|
| 813 |
+
if len(groups) >= 2:
|
| 814 |
+
tail_length = sum(len(item["text"]) for item in groups[-1])
|
| 815 |
+
tail_length += 2 * (len(groups[-1]) - 1)
|
| 816 |
+
previous_length = sum(len(item["text"]) for item in groups[-2])
|
| 817 |
+
previous_length += 2 * (len(groups[-2]) - 1)
|
| 818 |
+
if tail_length < min_chars and previous_length + 2 + tail_length <= max_chars:
|
| 819 |
+
groups[-2].extend(groups.pop())
|
| 820 |
+
|
| 821 |
+
result: list[dict[str, Any]] = []
|
| 822 |
+
for index, group in enumerate(groups):
|
| 823 |
+
# The most substantial section is the best label when a chunk spans a run
|
| 824 |
+
# of short sections. Every original heading remains embedded in the text.
|
| 825 |
+
anchor = max(group, key=lambda item: len(item["text"]))
|
| 826 |
+
identity = f"{canonical_url}\n{anchor['key']}"
|
| 827 |
+
chunk_id = f"tai-{_sha256(identity)[:24]}"
|
| 828 |
+
result.append(
|
| 829 |
+
{
|
| 830 |
+
"chunk_id": chunk_id,
|
| 831 |
+
"text": "\n\n".join(item["text"] for item in group),
|
| 832 |
+
"heading": anchor["heading"],
|
| 833 |
+
"index": index,
|
| 834 |
+
"evidence_spans": _span_records(
|
| 835 |
+
canonical_url,
|
| 836 |
+
chunk_id,
|
| 837 |
+
(span for item in group for span in item.get("spans", [])),
|
| 838 |
+
),
|
| 839 |
+
}
|
| 840 |
+
)
|
| 841 |
+
return result
|
| 842 |
+
|
| 843 |
+
|
| 844 |
+
def _document_text(sections: Iterable[dict[str, str]]) -> str:
|
| 845 |
+
blocks: list[str] = []
|
| 846 |
+
for section in sections:
|
| 847 |
+
heading = _clean(str(section.get("heading", "")))
|
| 848 |
+
body = _clean(str(section.get("text", "")))
|
| 849 |
+
block = f"{heading}\n{body}".strip()
|
| 850 |
+
if block:
|
| 851 |
+
blocks.append(block)
|
| 852 |
+
return "\n\n".join(blocks)
|
| 853 |
+
|
| 854 |
+
|
| 855 |
+
def _parse_sitemap(response_text: str) -> tuple[str, list[dict[str, str]]]:
|
| 856 |
+
root = ElementTree.fromstring(response_text)
|
| 857 |
+
root_type = _local_name(root.tag)
|
| 858 |
+
entries: list[dict[str, str]] = []
|
| 859 |
+
for child in root:
|
| 860 |
+
if _local_name(child.tag) not in {"url", "sitemap"}:
|
| 861 |
+
continue
|
| 862 |
+
values = {_local_name(item.tag): _clean(item.text or "") for item in child}
|
| 863 |
+
if values.get("loc"):
|
| 864 |
+
entries.append(values)
|
| 865 |
+
return root_type, entries
|
| 866 |
+
|
| 867 |
+
|
| 868 |
+
def discover_sitemap_entries(
|
| 869 |
+
session: requests.Session,
|
| 870 |
+
sitemap_url: str,
|
| 871 |
+
*,
|
| 872 |
+
allowed_hosts: frozenset[str] | None = None,
|
| 873 |
+
_visited: set[str] | None = None,
|
| 874 |
+
) -> list[dict[str, str]]:
|
| 875 |
+
"""Read a urlset or sitemap index and return de-duplicated page entries."""
|
| 876 |
+
|
| 877 |
+
visited = _visited if _visited is not None else set()
|
| 878 |
+
normalised_sitemap = _normalise_url(sitemap_url)
|
| 879 |
+
if normalised_sitemap in visited:
|
| 880 |
+
return []
|
| 881 |
+
visited.add(normalised_sitemap)
|
| 882 |
+
|
| 883 |
+
response = session.get(sitemap_url, timeout=30)
|
| 884 |
+
response.raise_for_status()
|
| 885 |
+
root_type, raw_entries = _parse_sitemap(response.text)
|
| 886 |
+
if root_type == "sitemapindex":
|
| 887 |
+
nested: list[dict[str, str]] = []
|
| 888 |
+
for item in raw_entries:
|
| 889 |
+
nested.extend(
|
| 890 |
+
discover_sitemap_entries(
|
| 891 |
+
session,
|
| 892 |
+
item["loc"],
|
| 893 |
+
allowed_hosts=allowed_hosts,
|
| 894 |
+
_visited=visited,
|
| 895 |
+
)
|
| 896 |
+
)
|
| 897 |
+
return nested
|
| 898 |
+
if root_type != "urlset":
|
| 899 |
+
raise ValueError(f"unsupported sitemap root: {root_type}")
|
| 900 |
+
|
| 901 |
+
result: list[dict[str, str]] = []
|
| 902 |
+
seen: set[str] = set()
|
| 903 |
+
for item in raw_entries:
|
| 904 |
+
url = _normalise_url(item["loc"])
|
| 905 |
+
parsed = urlparse(url)
|
| 906 |
+
if parsed.scheme not in {"http", "https"} or not parsed.hostname:
|
| 907 |
+
continue
|
| 908 |
+
if allowed_hosts and parsed.hostname.lower() not in allowed_hosts:
|
| 909 |
+
continue
|
| 910 |
+
if url in seen:
|
| 911 |
+
continue
|
| 912 |
+
seen.add(url)
|
| 913 |
+
result.append(
|
| 914 |
+
{
|
| 915 |
+
"url": url,
|
| 916 |
+
"lastmod": item.get("lastmod", ""),
|
| 917 |
+
"sitemap_url": normalised_sitemap,
|
| 918 |
+
}
|
| 919 |
+
)
|
| 920 |
+
return result
|
| 921 |
+
|
| 922 |
+
|
| 923 |
+
def discover_sitemap_urls(
|
| 924 |
+
session: requests.Session,
|
| 925 |
+
sitemap_url: str,
|
| 926 |
+
*,
|
| 927 |
+
allowed_hosts: frozenset[str] | None = None,
|
| 928 |
+
) -> list[str]:
|
| 929 |
+
return [
|
| 930 |
+
item["url"]
|
| 931 |
+
for item in discover_sitemap_entries(
|
| 932 |
+
session, sitemap_url, allowed_hosts=allowed_hosts
|
| 933 |
+
)
|
| 934 |
+
]
|
| 935 |
+
|
| 936 |
+
|
| 937 |
+
def merge_curated_com_metadata(
|
| 938 |
+
entries: Iterable[dict[str, str]],
|
| 939 |
+
curated_pages: Iterable[dict[str, str]] = CURATED_COM_METADATA,
|
| 940 |
+
) -> list[dict[str, str]]:
|
| 941 |
+
"""Attach non-factual labels only to URLs discovered in the live sitemap."""
|
| 942 |
+
|
| 943 |
+
curated_by_url = {_normalise_url(spec["url"]): dict(spec) for spec in curated_pages}
|
| 944 |
+
result: list[dict[str, str]] = []
|
| 945 |
+
for raw_entry in entries:
|
| 946 |
+
entry = dict(raw_entry)
|
| 947 |
+
url = _normalise_url(entry["url"])
|
| 948 |
+
entry["url"] = url
|
| 949 |
+
metadata = curated_by_url.get(url)
|
| 950 |
+
if metadata:
|
| 951 |
+
entry.update(
|
| 952 |
+
{key: value for key, value in metadata.items() if key != "url"}
|
| 953 |
+
)
|
| 954 |
+
result.append(entry)
|
| 955 |
+
return result
|
| 956 |
+
|
| 957 |
+
|
| 958 |
+
def _fetch_and_resolve(
|
| 959 |
+
session: requests.Session, url: str, *, max_meta_refreshes: int = 3
|
| 960 |
+
) -> tuple[Any, PageParser, str, list[str]]:
|
| 961 |
+
"""Follow HTTP redirects, HTML meta refreshes, then honour rel=canonical."""
|
| 962 |
+
|
| 963 |
+
current_url = url
|
| 964 |
+
visited: list[str] = []
|
| 965 |
+
for _hop in range(max_meta_refreshes + 1):
|
| 966 |
+
response = session.get(current_url, timeout=30)
|
| 967 |
+
response.raise_for_status()
|
| 968 |
+
response_url = _normalise_url(response.url)
|
| 969 |
+
visited.append(response_url)
|
| 970 |
+
parser = PageParser(response_url)
|
| 971 |
+
parser.feed(response.text)
|
| 972 |
+
parser.close()
|
| 973 |
+
|
| 974 |
+
refresh = parser.meta_refresh_url
|
| 975 |
+
if refresh:
|
| 976 |
+
refresh_url = _normalise_url(urljoin(response_url, refresh))
|
| 977 |
+
if refresh_url not in visited:
|
| 978 |
+
current_url = refresh_url
|
| 979 |
+
continue
|
| 980 |
+
|
| 981 |
+
canonical_url = response_url
|
| 982 |
+
if parser.canonical_url:
|
| 983 |
+
candidate = _normalise_url(urljoin(response_url, parser.canonical_url))
|
| 984 |
+
if urlparse(candidate).scheme in {"http", "https"}:
|
| 985 |
+
canonical_url = candidate
|
| 986 |
+
return response, parser, canonical_url, visited
|
| 987 |
+
raise ValueError(f"too many meta refreshes while fetching {url}")
|
| 988 |
+
|
| 989 |
+
|
| 990 |
+
def classify_kind(url: str) -> str:
|
| 991 |
+
path = _path(url).lower()
|
| 992 |
+
if path in {"/academy", "/academy/bundles"} or path.startswith("/collections"):
|
| 993 |
+
return "collection"
|
| 994 |
+
if "/enterprise/" in path or path in {
|
| 995 |
+
"/enterpriseenablement",
|
| 996 |
+
"/valuecreation",
|
| 997 |
+
}:
|
| 998 |
+
return "b2b"
|
| 999 |
+
if path.startswith("/valuecreation/"):
|
| 1000 |
+
return "career"
|
| 1001 |
+
if "/mentorship" in path or path.endswith("/tai-mentorship"):
|
| 1002 |
+
return "mentorship"
|
| 1003 |
+
if "free" in path or "/webinars/" in path:
|
| 1004 |
+
return "free_resource"
|
| 1005 |
+
if "/bundles/" in path:
|
| 1006 |
+
return "bundle"
|
| 1007 |
+
if "/courses/" in path or (
|
| 1008 |
+
path.startswith("/academy/")
|
| 1009 |
+
and any(
|
| 1010 |
+
term in path
|
| 1011 |
+
for term in (
|
| 1012 |
+
"full-stack",
|
| 1013 |
+
"agent-engineering",
|
| 1014 |
+
"llm-primer",
|
| 1015 |
+
"python-for-ai",
|
| 1016 |
+
"ai-for-work",
|
| 1017 |
+
"building-llms",
|
| 1018 |
+
)
|
| 1019 |
+
)
|
| 1020 |
+
):
|
| 1021 |
+
return "course"
|
| 1022 |
+
if "/digital_downloads/" in path:
|
| 1023 |
+
return "digital_download"
|
| 1024 |
+
if path.endswith("/book"):
|
| 1025 |
+
return "book"
|
| 1026 |
+
if path == "/contribute":
|
| 1027 |
+
return "community"
|
| 1028 |
+
return "page"
|
| 1029 |
+
|
| 1030 |
+
|
| 1031 |
+
def offer_id_for_url(url: str) -> str | None:
|
| 1032 |
+
return OFFER_PATHS.get(_path(url).lower())
|
| 1033 |
+
|
| 1034 |
+
|
| 1035 |
+
def entity_id_for_url(url: str) -> str:
|
| 1036 |
+
offer_id = offer_id_for_url(url)
|
| 1037 |
+
if offer_id:
|
| 1038 |
+
return f"offer:{offer_id}"
|
| 1039 |
+
parsed = urlparse(url)
|
| 1040 |
+
host = (parsed.hostname or "").lower()
|
| 1041 |
+
return f"page:{host}{_path(url).lower()}"
|
| 1042 |
+
|
| 1043 |
+
|
| 1044 |
+
def fetch_page(
|
| 1045 |
+
session: requests.Session,
|
| 1046 |
+
spec: dict[str, Any],
|
| 1047 |
+
*,
|
| 1048 |
+
authority: str | None = None,
|
| 1049 |
+
fetched_at: str | None = None,
|
| 1050 |
+
allowed_hosts: frozenset[str] | None = None,
|
| 1051 |
+
) -> dict[str, Any]:
|
| 1052 |
+
"""Fetch one sitemap entry without adding any non-page summary to its text."""
|
| 1053 |
+
|
| 1054 |
+
discovered_url = _normalise_url(str(spec["url"]))
|
| 1055 |
+
fetched_at = fetched_at or _utc_now()
|
| 1056 |
+
authority = authority or (
|
| 1057 |
+
"official_academy"
|
| 1058 |
+
if urlparse(discovered_url).hostname == "academy.towardsai.net"
|
| 1059 |
+
else "official_site"
|
| 1060 |
+
)
|
| 1061 |
+
response, parser, canonical_url, redirect_chain = _fetch_and_resolve(
|
| 1062 |
+
session, discovered_url
|
| 1063 |
+
)
|
| 1064 |
+
parsed = urlparse(canonical_url)
|
| 1065 |
+
text = _document_text(parser.sections)
|
| 1066 |
+
content_sha256 = _sha256(text)
|
| 1067 |
+
status = "included"
|
| 1068 |
+
excluded_reason = ""
|
| 1069 |
+
if allowed_hosts and (parsed.hostname or "").lower() not in allowed_hosts:
|
| 1070 |
+
status = "excluded"
|
| 1071 |
+
excluded_reason = "canonical URL is outside the sitemap authority"
|
| 1072 |
+
|
| 1073 |
+
discovered_path = _path(discovered_url)
|
| 1074 |
+
if authority == "official_site" and discovered_path in COM_EXCLUSIONS:
|
| 1075 |
+
status = "excluded"
|
| 1076 |
+
excluded_reason = COM_EXCLUSIONS[discovered_path]
|
| 1077 |
+
if authority == "official_academy" and discovered_path in ACADEMY_EXCLUSIONS:
|
| 1078 |
+
status = "excluded"
|
| 1079 |
+
excluded_reason = ACADEMY_EXCLUSIONS[discovered_path]
|
| 1080 |
+
if not text:
|
| 1081 |
+
status = "excluded"
|
| 1082 |
+
excluded_reason = "no usable visible page content"
|
| 1083 |
+
|
| 1084 |
+
title = _clean(" ".join(parser.title_parts))
|
| 1085 |
+
if not title:
|
| 1086 |
+
title = _clean(str(spec.get("review_title", ""))) or canonical_url
|
| 1087 |
+
offer_id = offer_id_for_url(canonical_url) or offer_id_for_url(discovered_url)
|
| 1088 |
+
chunks = build_chunks(parser.sections, canonical_url)
|
| 1089 |
+
page = {
|
| 1090 |
+
"discovered_url": discovered_url,
|
| 1091 |
+
"url": canonical_url,
|
| 1092 |
+
"canonical_url": canonical_url,
|
| 1093 |
+
"host": (parsed.hostname or "").lower(),
|
| 1094 |
+
"path": parsed.path.rstrip("/") or "/",
|
| 1095 |
+
"kind": str(spec.get("kind") or classify_kind(canonical_url)),
|
| 1096 |
+
"offer_id": offer_id,
|
| 1097 |
+
"entity_id": (
|
| 1098 |
+
f"offer:{offer_id}" if offer_id else entity_id_for_url(canonical_url)
|
| 1099 |
+
),
|
| 1100 |
+
"title": title,
|
| 1101 |
+
"review_title": str(spec.get("review_title", "")),
|
| 1102 |
+
"reviewed_summary": str(spec.get("reviewed_summary", "")),
|
| 1103 |
+
"meta_description": parser.meta_description,
|
| 1104 |
+
"headings": list(dict.fromkeys(parser.headings))[:100],
|
| 1105 |
+
"text": text,
|
| 1106 |
+
"links": _unique_links(parser.links)[:300],
|
| 1107 |
+
"chunks": chunks,
|
| 1108 |
+
"fetched_at": fetched_at,
|
| 1109 |
+
"content_sha256": content_sha256,
|
| 1110 |
+
# content_hash is the generic name consumed by the retrieval layer.
|
| 1111 |
+
"content_hash": content_sha256,
|
| 1112 |
+
"authority": authority,
|
| 1113 |
+
"status": status,
|
| 1114 |
+
"retrieval_eligible": status == "included",
|
| 1115 |
+
"http_status": int(getattr(response, "status_code", 200)),
|
| 1116 |
+
"lastmod": str(spec.get("lastmod", "")),
|
| 1117 |
+
"sitemap_url": str(spec.get("sitemap_url", "")),
|
| 1118 |
+
"redirect_chain": redirect_chain,
|
| 1119 |
+
}
|
| 1120 |
+
if excluded_reason:
|
| 1121 |
+
page["excluded_reason"] = excluded_reason
|
| 1122 |
+
page["evidence_hash"] = _evidence_hash(page)
|
| 1123 |
+
return page
|
| 1124 |
+
|
| 1125 |
+
|
| 1126 |
+
def _failed_page(
|
| 1127 |
+
spec: dict[str, Any], authority: str, fetched_at: str, error: Exception
|
| 1128 |
+
) -> dict[str, Any]:
|
| 1129 |
+
url = _normalise_url(str(spec["url"]))
|
| 1130 |
+
parsed = urlparse(url)
|
| 1131 |
+
content_sha256 = _sha256("")
|
| 1132 |
+
chunks: list[dict[str, Any]] = []
|
| 1133 |
+
page = {
|
| 1134 |
+
"discovered_url": url,
|
| 1135 |
+
"url": url,
|
| 1136 |
+
"canonical_url": url,
|
| 1137 |
+
"host": (parsed.hostname or "").lower(),
|
| 1138 |
+
"path": parsed.path.rstrip("/") or "/",
|
| 1139 |
+
"kind": classify_kind(url),
|
| 1140 |
+
"offer_id": offer_id_for_url(url),
|
| 1141 |
+
"entity_id": entity_id_for_url(url),
|
| 1142 |
+
"title": url,
|
| 1143 |
+
"review_title": str(spec.get("review_title", "")),
|
| 1144 |
+
"reviewed_summary": str(spec.get("reviewed_summary", "")),
|
| 1145 |
+
"meta_description": "",
|
| 1146 |
+
"headings": [],
|
| 1147 |
+
"text": "",
|
| 1148 |
+
"links": [],
|
| 1149 |
+
"chunks": chunks,
|
| 1150 |
+
"fetched_at": fetched_at,
|
| 1151 |
+
"content_sha256": content_sha256,
|
| 1152 |
+
"content_hash": content_sha256,
|
| 1153 |
+
"authority": authority,
|
| 1154 |
+
"status": "fetch_error",
|
| 1155 |
+
"retrieval_eligible": False,
|
| 1156 |
+
"http_status": None,
|
| 1157 |
+
"lastmod": str(spec.get("lastmod", "")),
|
| 1158 |
+
"sitemap_url": str(spec.get("sitemap_url", "")),
|
| 1159 |
+
"redirect_chain": [],
|
| 1160 |
+
"error": f"{type(error).__name__}: {error}",
|
| 1161 |
+
}
|
| 1162 |
+
page["evidence_hash"] = _evidence_hash(page)
|
| 1163 |
+
return page
|
| 1164 |
+
|
| 1165 |
+
|
| 1166 |
+
def _exclude_canonical_duplicates(pages: list[dict[str, Any]]) -> None:
|
| 1167 |
+
seen: dict[str, str] = {}
|
| 1168 |
+
for page in pages:
|
| 1169 |
+
if page["status"] != "included":
|
| 1170 |
+
continue
|
| 1171 |
+
canonical_url = str(page["canonical_url"])
|
| 1172 |
+
if canonical_url in seen:
|
| 1173 |
+
page["status"] = "excluded"
|
| 1174 |
+
page["retrieval_eligible"] = False
|
| 1175 |
+
page["excluded_reason"] = (
|
| 1176 |
+
f"duplicate canonical URL; authoritative scan is {seen[canonical_url]}"
|
| 1177 |
+
)
|
| 1178 |
+
else:
|
| 1179 |
+
seen[canonical_url] = str(page["discovered_url"])
|
| 1180 |
+
|
| 1181 |
+
|
| 1182 |
+
def _manual_resource_page(spec: dict[str, Any], fetched_at: str) -> dict[str, Any]:
|
| 1183 |
+
url = _normalise_url(str(spec["url"]))
|
| 1184 |
+
parsed = urlparse(url)
|
| 1185 |
+
text = _clean(str(spec["text"]))
|
| 1186 |
+
content_sha256 = _sha256(text)
|
| 1187 |
+
chunks: list[dict[str, Any]] = []
|
| 1188 |
+
page = {
|
| 1189 |
+
"discovered_url": url,
|
| 1190 |
+
"url": url,
|
| 1191 |
+
"canonical_url": url,
|
| 1192 |
+
"host": (parsed.hostname or "").lower(),
|
| 1193 |
+
"path": parsed.path.rstrip("/") or "/",
|
| 1194 |
+
"kind": str(spec["kind"]),
|
| 1195 |
+
"offer_id": None,
|
| 1196 |
+
"entity_id": entity_id_for_url(url),
|
| 1197 |
+
"title": str(spec["title"]),
|
| 1198 |
+
"meta_description": "",
|
| 1199 |
+
"headings": [str(spec["title"])],
|
| 1200 |
+
"text": text,
|
| 1201 |
+
"links": list(spec.get("links", [])),
|
| 1202 |
+
"chunks": chunks,
|
| 1203 |
+
"fetched_at": fetched_at,
|
| 1204 |
+
"content_sha256": content_sha256,
|
| 1205 |
+
"content_hash": content_sha256,
|
| 1206 |
+
"authority": "curated_external",
|
| 1207 |
+
"status": "excluded",
|
| 1208 |
+
"retrieval_eligible": False,
|
| 1209 |
+
"excluded_reason": (
|
| 1210 |
+
"manual routing-only resource; descriptive text was not fetched"
|
| 1211 |
+
),
|
| 1212 |
+
"http_status": None,
|
| 1213 |
+
"lastmod": "",
|
| 1214 |
+
"sitemap_url": "manual_allowlist",
|
| 1215 |
+
"redirect_chain": [],
|
| 1216 |
+
}
|
| 1217 |
+
page["evidence_hash"] = _evidence_hash(page)
|
| 1218 |
+
return page
|
| 1219 |
+
|
| 1220 |
+
|
| 1221 |
+
def build_catalog(
|
| 1222 |
+
*,
|
| 1223 |
+
session: requests.Session | None = None,
|
| 1224 |
+
sitemap_url: str = COM_SITEMAP_URL,
|
| 1225 |
+
authority: str = "official_site",
|
| 1226 |
+
allowed_hosts: frozenset[str] = COM_HOSTS,
|
| 1227 |
+
fetched_at: str | None = None,
|
| 1228 |
+
include_manual_resources: bool = False,
|
| 1229 |
+
) -> dict[str, Any]:
|
| 1230 |
+
"""Build one catalog from its sitemap while retaining failed scan records."""
|
| 1231 |
+
|
| 1232 |
+
fetched_at = fetched_at or _utc_now()
|
| 1233 |
+
own_session = session or requests.Session()
|
| 1234 |
+
own_session.headers.setdefault("User-Agent", "TowardsAIHelperCatalog/2.0")
|
| 1235 |
+
entries = discover_sitemap_entries(
|
| 1236 |
+
own_session, sitemap_url, allowed_hosts=allowed_hosts
|
| 1237 |
+
)
|
| 1238 |
+
if authority == "official_site":
|
| 1239 |
+
entries = merge_curated_com_metadata(entries)
|
| 1240 |
+
pages: list[dict[str, Any]] = []
|
| 1241 |
+
for spec in entries:
|
| 1242 |
+
try:
|
| 1243 |
+
pages.append(
|
| 1244 |
+
fetch_page(
|
| 1245 |
+
own_session,
|
| 1246 |
+
spec,
|
| 1247 |
+
authority=authority,
|
| 1248 |
+
fetched_at=fetched_at,
|
| 1249 |
+
allowed_hosts=allowed_hosts,
|
| 1250 |
+
)
|
| 1251 |
+
)
|
| 1252 |
+
except (requests.RequestException, ValueError) as error:
|
| 1253 |
+
pages.append(_failed_page(spec, authority, fetched_at, error))
|
| 1254 |
+
_exclude_canonical_duplicates(pages)
|
| 1255 |
+
if include_manual_resources:
|
| 1256 |
+
pages.extend(
|
| 1257 |
+
_manual_resource_page(spec, fetched_at) for spec in MANUAL_SAFE_RESOURCES
|
| 1258 |
+
)
|
| 1259 |
+
for page in pages:
|
| 1260 |
+
page["evidence_hash"] = _evidence_hash(page)
|
| 1261 |
+
|
| 1262 |
+
statuses: dict[str, int] = {}
|
| 1263 |
+
for page in pages:
|
| 1264 |
+
status = str(page["status"])
|
| 1265 |
+
statuses[status] = statuses.get(status, 0) + 1
|
| 1266 |
+
return {
|
| 1267 |
+
"schema_version": 2,
|
| 1268 |
+
"source": _normalise_url(sitemap_url),
|
| 1269 |
+
"generated_at": fetched_at,
|
| 1270 |
+
"authority": authority,
|
| 1271 |
+
"status_counts": statuses,
|
| 1272 |
+
"pages": pages,
|
| 1273 |
+
}
|
| 1274 |
+
|
| 1275 |
+
|
| 1276 |
+
def build_catalogs(
|
| 1277 |
+
*,
|
| 1278 |
+
session: requests.Session | None = None,
|
| 1279 |
+
com_sitemap_url: str = COM_SITEMAP_URL,
|
| 1280 |
+
academy_sitemap_url: str = ACADEMY_SITEMAP_URL,
|
| 1281 |
+
fetched_at: str | None = None,
|
| 1282 |
+
) -> tuple[dict[str, Any], dict[str, Any]]:
|
| 1283 |
+
"""Build the public-site and Academy catalogs in one reproducible scan."""
|
| 1284 |
+
|
| 1285 |
+
fetched_at = fetched_at or _utc_now()
|
| 1286 |
+
own_session = session or requests.Session()
|
| 1287 |
+
own_session.headers.setdefault("User-Agent", "TowardsAIHelperCatalog/2.0")
|
| 1288 |
+
com_catalog = build_catalog(
|
| 1289 |
+
session=own_session,
|
| 1290 |
+
sitemap_url=com_sitemap_url,
|
| 1291 |
+
authority="official_site",
|
| 1292 |
+
allowed_hosts=COM_HOSTS,
|
| 1293 |
+
fetched_at=fetched_at,
|
| 1294 |
+
)
|
| 1295 |
+
academy_catalog = build_catalog(
|
| 1296 |
+
session=own_session,
|
| 1297 |
+
sitemap_url=academy_sitemap_url,
|
| 1298 |
+
authority="official_academy",
|
| 1299 |
+
allowed_hosts=ACADEMY_HOSTS,
|
| 1300 |
+
fetched_at=fetched_at,
|
| 1301 |
+
include_manual_resources=True,
|
| 1302 |
+
)
|
| 1303 |
+
return com_catalog, academy_catalog
|
| 1304 |
+
|
| 1305 |
+
|
| 1306 |
+
def _write_catalog(output: Path, payload: dict[str, Any]) -> None:
|
| 1307 |
+
output.parent.mkdir(parents=True, exist_ok=True)
|
| 1308 |
+
output.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n")
|
| 1309 |
+
|
| 1310 |
+
|
| 1311 |
+
def main() -> None:
|
| 1312 |
+
parser = argparse.ArgumentParser(
|
| 1313 |
+
description="Refresh the towardsai.com and Academy helper catalogs"
|
| 1314 |
+
)
|
| 1315 |
+
parser.add_argument(
|
| 1316 |
+
"--com-output",
|
| 1317 |
+
"--output",
|
| 1318 |
+
dest="com_output",
|
| 1319 |
+
type=Path,
|
| 1320 |
+
default=DEFAULT_COM_OUTPUT,
|
| 1321 |
+
help="output for towardsai.com pages (legacy alias: --output)",
|
| 1322 |
+
)
|
| 1323 |
+
parser.add_argument("--academy-output", type=Path, default=DEFAULT_ACADEMY_OUTPUT)
|
| 1324 |
+
parser.add_argument("--com-sitemap", default=COM_SITEMAP_URL)
|
| 1325 |
+
parser.add_argument("--academy-sitemap", default=ACADEMY_SITEMAP_URL)
|
| 1326 |
+
args = parser.parse_args()
|
| 1327 |
+
|
| 1328 |
+
com_catalog, academy_catalog = build_catalogs(
|
| 1329 |
+
com_sitemap_url=args.com_sitemap,
|
| 1330 |
+
academy_sitemap_url=args.academy_sitemap,
|
| 1331 |
+
)
|
| 1332 |
+
_write_catalog(args.com_output, com_catalog)
|
| 1333 |
+
_write_catalog(args.academy_output, academy_catalog)
|
| 1334 |
+
print(f"Wrote {len(com_catalog['pages'])} .com pages to {args.com_output}")
|
| 1335 |
+
print(
|
| 1336 |
+
f"Wrote {len(academy_catalog['pages'])} Academy/manual pages "
|
| 1337 |
+
f"to {args.academy_output}"
|
| 1338 |
+
)
|
| 1339 |
+
|
| 1340 |
+
|
| 1341 |
+
if __name__ == "__main__":
|
| 1342 |
+
main()
|
static/widget.js
CHANGED
|
@@ -16,6 +16,9 @@
|
|
| 16 |
}
|
| 17 |
if (!apiBase) return;
|
| 18 |
|
|
|
|
|
|
|
|
|
|
| 19 |
var state = {
|
| 20 |
config: null,
|
| 21 |
visible: false,
|
|
@@ -268,15 +271,28 @@
|
|
| 268 |
|
| 269 |
function isBlockedPath() {
|
| 270 |
var path = normalizePath(window.location.pathname);
|
| 271 |
-
|
| 272 |
-
|
| 273 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 274 |
}
|
| 275 |
|
| 276 |
function pageAllowed(config) {
|
| 277 |
if (!config || isSignedIn() || isBlockedPath()) return false;
|
| 278 |
var host = window.location.hostname.toLowerCase();
|
| 279 |
var path = normalizePath(window.location.pathname);
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 280 |
var paths = config.allowedPathsByHost && config.allowedPathsByHost[host];
|
| 281 |
if (!paths && host.indexOf("www.") === 0) {
|
| 282 |
paths = config.allowedPathsByHost && config.allowedPathsByHost[host.slice(4)];
|
|
@@ -370,9 +386,17 @@
|
|
| 370 |
.then(function (response) {
|
| 371 |
if (!response.ok) {
|
| 372 |
if (response.status === 429) {
|
| 373 |
-
throw new Error(
|
|
|
|
|
|
|
|
|
|
|
|
|
| 374 |
}
|
| 375 |
-
throw new Error(
|
|
|
|
|
|
|
|
|
|
|
|
|
| 376 |
}
|
| 377 |
return response.json();
|
| 378 |
})
|
|
@@ -401,7 +425,10 @@
|
|
| 401 |
})
|
| 402 |
.catch(function (error) {
|
| 403 |
loading.classList.remove("empty");
|
| 404 |
-
loading.innerHTML = renderMarkdown(
|
|
|
|
|
|
|
|
|
|
| 405 |
})
|
| 406 |
.finally(function () {
|
| 407 |
setBusy(false);
|
|
|
|
| 16 |
}
|
| 17 |
if (!apiBase) return;
|
| 18 |
|
| 19 |
+
var contactFormUrl = "https://towardsai.com/academy/contact/#contact";
|
| 20 |
+
var contactLinkMarkdown = "[contact the Towards AI team](" + contactFormUrl + ")";
|
| 21 |
+
|
| 22 |
var state = {
|
| 23 |
config: null,
|
| 24 |
visible: false,
|
|
|
|
| 271 |
|
| 272 |
function isBlockedPath() {
|
| 273 |
var path = normalizePath(window.location.pathname);
|
| 274 |
+
var privatePath = [
|
| 275 |
+
"courses/take", "enroll", "order", "checkout", "cart", "users", "account",
|
| 276 |
+
"admin", "wp-admin", "wp-login", "wp-json", "wp-content", "wp-includes",
|
| 277 |
+
"xmlrpc.php",
|
| 278 |
+
].some(function (prefix) {
|
| 279 |
+
return path.replace(/^\//, "").indexOf(prefix) === 0;
|
| 280 |
+
});
|
| 281 |
+
return privatePath || new URLSearchParams(window.location.search).has("preview");
|
| 282 |
}
|
| 283 |
|
| 284 |
function pageAllowed(config) {
|
| 285 |
if (!config || isSignedIn() || isBlockedPath()) return false;
|
| 286 |
var host = window.location.hostname.toLowerCase();
|
| 287 |
var path = normalizePath(window.location.pathname);
|
| 288 |
+
if (
|
| 289 |
+
Array.isArray(config.siteWideHosts) &&
|
| 290 |
+
config.siteWideHosts
|
| 291 |
+
.map(function (item) { return item.toLowerCase(); })
|
| 292 |
+
.indexOf(host) !== -1
|
| 293 |
+
) {
|
| 294 |
+
return true;
|
| 295 |
+
}
|
| 296 |
var paths = config.allowedPathsByHost && config.allowedPathsByHost[host];
|
| 297 |
if (!paths && host.indexOf("www.") === 0) {
|
| 298 |
paths = config.allowedPathsByHost && config.allowedPathsByHost[host.slice(4)];
|
|
|
|
| 386 |
.then(function (response) {
|
| 387 |
if (!response.ok) {
|
| 388 |
if (response.status === 429) {
|
| 389 |
+
throw new Error(
|
| 390 |
+
"The helper is rate limited right now. Please try again later or " +
|
| 391 |
+
contactLinkMarkdown +
|
| 392 |
+
"."
|
| 393 |
+
);
|
| 394 |
}
|
| 395 |
+
throw new Error(
|
| 396 |
+
"The helper is unavailable on this page. Please " +
|
| 397 |
+
contactLinkMarkdown +
|
| 398 |
+
"."
|
| 399 |
+
);
|
| 400 |
}
|
| 401 |
return response.json();
|
| 402 |
})
|
|
|
|
| 425 |
})
|
| 426 |
.catch(function (error) {
|
| 427 |
loading.classList.remove("empty");
|
| 428 |
+
loading.innerHTML = renderMarkdown(
|
| 429 |
+
error.message ||
|
| 430 |
+
"Something went wrong. Please " + contactLinkMarkdown + "."
|
| 431 |
+
);
|
| 432 |
})
|
| 433 |
.finally(function () {
|
| 434 |
setBusy(false);
|
tai_helper/api.py
CHANGED
|
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|
| 2 |
|
| 3 |
import asyncio
|
| 4 |
import logging
|
|
|
|
| 5 |
from typing import Any
|
| 6 |
from urllib.parse import urlparse
|
| 7 |
from uuid import uuid4
|
|
@@ -16,6 +17,7 @@ from .catalog import (
|
|
| 16 |
allowed_paths_by_host,
|
| 17 |
coupon_followup,
|
| 18 |
coupon_intent,
|
|
|
|
| 19 |
forced_prompts,
|
| 20 |
in_scope,
|
| 21 |
page_is_allowed,
|
|
@@ -29,6 +31,8 @@ from .settings import repo_root, settings
|
|
| 29 |
|
| 30 |
logger = logging.getLogger(__name__)
|
| 31 |
|
|
|
|
|
|
|
| 32 |
app = FastAPI(title="Towards AI Helper API", version="0.1.0")
|
| 33 |
app.add_middleware(
|
| 34 |
CORSMiddleware,
|
|
@@ -140,11 +144,64 @@ def _history_text(payload: HelperChatRequest) -> list[str]:
|
|
| 140 |
return [turn.content for turn in payload.history[-settings.max_history_turns :]]
|
| 141 |
|
| 142 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 143 |
def _validate_payload(payload: HelperChatRequest) -> None:
|
| 144 |
if payload.context.signedIn:
|
| 145 |
-
raise HTTPException(
|
|
|
|
|
|
|
| 146 |
if not page_is_allowed(payload.context.url):
|
| 147 |
-
raise HTTPException(
|
|
|
|
|
|
|
| 148 |
if len(payload.query.strip()) > settings.max_query_chars:
|
| 149 |
raise HTTPException(status_code=400, detail="Question is too long.")
|
| 150 |
if not payload.history and payload.query.strip() not in forced_prompts():
|
|
@@ -159,14 +216,38 @@ def _fixed_coupon_answer(payload: HelperChatRequest) -> str:
|
|
| 159 |
if coupon_followup(payload.query, history):
|
| 160 |
return (
|
| 161 |
"I can't provide a coupon code here. If you have specific context, "
|
| 162 |
-
"
|
| 163 |
)
|
| 164 |
return (
|
| 165 |
-
"I can't provide a coupon code here.
|
| 166 |
-
"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 167 |
)
|
| 168 |
|
| 169 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 170 |
def _out_of_scope_answer() -> str:
|
| 171 |
return (
|
| 172 |
"I can only help with choosing Towards AI courses, bundles, mentorship, "
|
|
@@ -175,10 +256,11 @@ def _out_of_scope_answer() -> str:
|
|
| 175 |
)
|
| 176 |
|
| 177 |
|
| 178 |
-
def
|
| 179 |
return (
|
| 180 |
-
"I
|
| 181 |
-
"
|
|
|
|
| 182 |
)
|
| 183 |
|
| 184 |
|
|
@@ -205,6 +287,7 @@ def healthcheck() -> dict[str, str]:
|
|
| 205 |
return {"status": "ok"}
|
| 206 |
|
| 207 |
|
|
|
|
| 208 |
@app.get("/widget.js")
|
| 209 |
def widget() -> FileResponse:
|
| 210 |
path = STATIC_DIR / "widget.js"
|
|
@@ -219,6 +302,7 @@ def public_config() -> dict[str, Any]:
|
|
| 219 |
return {
|
| 220 |
"name": "Towards AI Helper",
|
| 221 |
"allowedHosts": list(settings.allowed_hosts),
|
|
|
|
| 222 |
"allowedPathsByHost": allowed_paths_by_host(),
|
| 223 |
"forcedPrompts": forced_prompts(),
|
| 224 |
"rateLimits": {
|
|
@@ -240,37 +324,84 @@ async def chat(request: Request, payload: HelperChatRequest) -> HelperChatRespon
|
|
| 240 |
|
| 241 |
query = payload.query.strip()
|
| 242 |
thread_id = payload.threadId.strip() or uuid4().hex
|
| 243 |
-
selected_pages
|
| 244 |
-
sources =
|
| 245 |
usage: dict[str, Any] = {}
|
| 246 |
latency_ms = 0
|
| 247 |
error_message = ""
|
|
|
|
| 248 |
|
| 249 |
try:
|
| 250 |
-
if
|
|
|
|
|
|
|
|
|
|
| 251 |
answer = _fixed_coupon_answer(payload)
|
|
|
|
| 252 |
elif not in_scope(query, _history_text(payload)):
|
| 253 |
answer = _out_of_scope_answer()
|
|
|
|
| 254 |
else:
|
| 255 |
-
|
| 256 |
-
|
| 257 |
-
selected_prompt=payload.selectedPrompt or query,
|
| 258 |
-
current_url=payload.context.url,
|
| 259 |
-
page_title=payload.context.pageTitle,
|
| 260 |
-
history=[
|
| 261 |
-
(turn.role, turn.content)
|
| 262 |
-
for turn in payload.history[-settings.max_history_turns :]
|
| 263 |
-
],
|
| 264 |
-
selected_pages=selected_pages,
|
| 265 |
)
|
| 266 |
-
|
| 267 |
-
|
| 268 |
-
|
| 269 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 270 |
except Exception:
|
| 271 |
error_message = "helper generation failed"
|
| 272 |
logger.exception("helper generation failed")
|
| 273 |
-
answer =
|
|
|
|
|
|
|
| 274 |
|
| 275 |
monitor = HelperMonitor(
|
| 276 |
query=query,
|
|
@@ -290,5 +421,6 @@ async def chat(request: Request, payload: HelperChatRequest) -> HelperChatRespon
|
|
| 290 |
answer=answer,
|
| 291 |
threadId=thread_id,
|
| 292 |
sources=[SourceOut(**source) for source in sources],
|
|
|
|
| 293 |
usage=usage,
|
| 294 |
)
|
|
|
|
| 2 |
|
| 3 |
import asyncio
|
| 4 |
import logging
|
| 5 |
+
import re
|
| 6 |
from typing import Any
|
| 7 |
from urllib.parse import urlparse
|
| 8 |
from uuid import uuid4
|
|
|
|
| 17 |
allowed_paths_by_host,
|
| 18 |
coupon_followup,
|
| 19 |
coupon_intent,
|
| 20 |
+
evidence_offer_ids_for_query,
|
| 21 |
forced_prompts,
|
| 22 |
in_scope,
|
| 23 |
page_is_allowed,
|
|
|
|
| 31 |
|
| 32 |
logger = logging.getLogger(__name__)
|
| 33 |
|
| 34 |
+
CONTACT_FORM_URL = "https://towardsai.com/academy/contact/#contact"
|
| 35 |
+
|
| 36 |
app = FastAPI(title="Towards AI Helper API", version="0.1.0")
|
| 37 |
app.add_middleware(
|
| 38 |
CORSMiddleware,
|
|
|
|
| 144 |
return [turn.content for turn in payload.history[-settings.max_history_turns :]]
|
| 145 |
|
| 146 |
|
| 147 |
+
def _user_history_text(payload: HelperChatRequest) -> list[str]:
|
| 148 |
+
return [
|
| 149 |
+
turn.content
|
| 150 |
+
for turn in payload.history[-settings.max_history_turns :]
|
| 151 |
+
if turn.role == "user"
|
| 152 |
+
]
|
| 153 |
+
|
| 154 |
+
|
| 155 |
+
def _retrieval_query(payload: HelperChatRequest) -> str:
|
| 156 |
+
"""Use current intent, adding only a real user referent when it is needed."""
|
| 157 |
+
|
| 158 |
+
query = payload.query.strip()
|
| 159 |
+
if not payload.history:
|
| 160 |
+
return query
|
| 161 |
+
|
| 162 |
+
words = set(re.findall(r"[a-z0-9'-]+", query.casefold()))
|
| 163 |
+
contextual_terms = {
|
| 164 |
+
"it",
|
| 165 |
+
"that",
|
| 166 |
+
"this",
|
| 167 |
+
"they",
|
| 168 |
+
"them",
|
| 169 |
+
"those",
|
| 170 |
+
"price",
|
| 171 |
+
"cost",
|
| 172 |
+
"discount",
|
| 173 |
+
"included",
|
| 174 |
+
"access",
|
| 175 |
+
"refund",
|
| 176 |
+
"duration",
|
| 177 |
+
"hours",
|
| 178 |
+
"lessons",
|
| 179 |
+
"prerequisites",
|
| 180 |
+
}
|
| 181 |
+
needs_context = len(words) <= 12 and bool(words & contextual_terms)
|
| 182 |
+
if not needs_context:
|
| 183 |
+
return query
|
| 184 |
+
|
| 185 |
+
starters = set(forced_prompts())
|
| 186 |
+
prior_user_turns = [
|
| 187 |
+
turn.strip()
|
| 188 |
+
for turn in _user_history_text(payload)
|
| 189 |
+
if turn.strip() and turn.strip() not in starters and turn.strip() != query
|
| 190 |
+
]
|
| 191 |
+
if not prior_user_turns:
|
| 192 |
+
return query
|
| 193 |
+
return f"{prior_user_turns[-1]}\n{query}"
|
| 194 |
+
|
| 195 |
+
|
| 196 |
def _validate_payload(payload: HelperChatRequest) -> None:
|
| 197 |
if payload.context.signedIn:
|
| 198 |
+
raise HTTPException(
|
| 199 |
+
status_code=403, detail="Helper is only for signed-out visitors."
|
| 200 |
+
)
|
| 201 |
if not page_is_allowed(payload.context.url):
|
| 202 |
+
raise HTTPException(
|
| 203 |
+
status_code=403, detail="Helper is only available on public pages."
|
| 204 |
+
)
|
| 205 |
if len(payload.query.strip()) > settings.max_query_chars:
|
| 206 |
raise HTTPException(status_code=400, detail="Question is too long.")
|
| 207 |
if not payload.history and payload.query.strip() not in forced_prompts():
|
|
|
|
| 216 |
if coupon_followup(payload.query, history):
|
| 217 |
return (
|
| 218 |
"I can't provide a coupon code here. If you have specific context, "
|
| 219 |
+
f"[contact the Towards AI team]({CONTACT_FORM_URL}) with what you need."
|
| 220 |
)
|
| 221 |
return (
|
| 222 |
+
"I can't provide a coupon code here. For pricing help based on your "
|
| 223 |
+
f"circumstances, [contact the Towards AI team]({CONTACT_FORM_URL})."
|
| 224 |
+
)
|
| 225 |
+
|
| 226 |
+
|
| 227 |
+
def _contact_intent(text: str) -> bool:
|
| 228 |
+
lowered = text.casefold()
|
| 229 |
+
return any(
|
| 230 |
+
phrase in lowered
|
| 231 |
+
for phrase in (
|
| 232 |
+
"contact form",
|
| 233 |
+
"contact the team",
|
| 234 |
+
"contact a person",
|
| 235 |
+
"contact someone",
|
| 236 |
+
"customer support",
|
| 237 |
+
"talk to a person",
|
| 238 |
+
"talk to someone",
|
| 239 |
+
"speak to a person",
|
| 240 |
+
"speak to someone",
|
| 241 |
+
"reach the team",
|
| 242 |
+
"email the team",
|
| 243 |
+
)
|
| 244 |
)
|
| 245 |
|
| 246 |
|
| 247 |
+
def _contact_answer() -> str:
|
| 248 |
+
return f"Please [contact the Towards AI team]({CONTACT_FORM_URL})."
|
| 249 |
+
|
| 250 |
+
|
| 251 |
def _out_of_scope_answer() -> str:
|
| 252 |
return (
|
| 253 |
"I can only help with choosing Towards AI courses, bundles, mentorship, "
|
|
|
|
| 256 |
)
|
| 257 |
|
| 258 |
|
| 259 |
+
def _insufficient_evidence_answer() -> str:
|
| 260 |
return (
|
| 261 |
+
"I couldn't verify that from the current Towards AI pages, so I don't want "
|
| 262 |
+
f"to guess. Please [contact the Towards AI team]({CONTACT_FORM_URL}) "
|
| 263 |
+
"for confirmation."
|
| 264 |
)
|
| 265 |
|
| 266 |
|
|
|
|
| 287 |
return {"status": "ok"}
|
| 288 |
|
| 289 |
|
| 290 |
+
@app.get("/helper-widget.js")
|
| 291 |
@app.get("/widget.js")
|
| 292 |
def widget() -> FileResponse:
|
| 293 |
path = STATIC_DIR / "widget.js"
|
|
|
|
| 302 |
return {
|
| 303 |
"name": "Towards AI Helper",
|
| 304 |
"allowedHosts": list(settings.allowed_hosts),
|
| 305 |
+
"siteWideHosts": list(settings.site_wide_hosts),
|
| 306 |
"allowedPathsByHost": allowed_paths_by_host(),
|
| 307 |
"forcedPrompts": forced_prompts(),
|
| 308 |
"rateLimits": {
|
|
|
|
| 324 |
|
| 325 |
query = payload.query.strip()
|
| 326 |
thread_id = payload.threadId.strip() or uuid4().hex
|
| 327 |
+
selected_pages: list[dict[str, Any]] = []
|
| 328 |
+
sources: list[dict[str, str]] = []
|
| 329 |
usage: dict[str, Any] = {}
|
| 330 |
latency_ms = 0
|
| 331 |
error_message = ""
|
| 332 |
+
response_status = "insufficient_evidence"
|
| 333 |
|
| 334 |
try:
|
| 335 |
+
if _contact_intent(query):
|
| 336 |
+
answer = _contact_answer()
|
| 337 |
+
response_status = "policy"
|
| 338 |
+
elif coupon_intent(query):
|
| 339 |
answer = _fixed_coupon_answer(payload)
|
| 340 |
+
response_status = "policy"
|
| 341 |
elif not in_scope(query, _history_text(payload)):
|
| 342 |
answer = _out_of_scope_answer()
|
| 343 |
+
response_status = "out_of_scope"
|
| 344 |
else:
|
| 345 |
+
selected_pages = retrieve(
|
| 346 |
+
_retrieval_query(payload), current_url=payload.context.url
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 347 |
)
|
| 348 |
+
if not selected_pages:
|
| 349 |
+
answer = _insufficient_evidence_answer()
|
| 350 |
+
else:
|
| 351 |
+
prompt = llm.build_prompt(
|
| 352 |
+
query=query,
|
| 353 |
+
selected_prompt=payload.selectedPrompt or query,
|
| 354 |
+
current_url=payload.context.url,
|
| 355 |
+
page_title=payload.context.pageTitle,
|
| 356 |
+
history=[
|
| 357 |
+
(turn.role, turn.content)
|
| 358 |
+
for turn in payload.history[-settings.max_history_turns :]
|
| 359 |
+
],
|
| 360 |
+
selected_pages=selected_pages,
|
| 361 |
+
)
|
| 362 |
+
retrieval_query = _retrieval_query(payload)
|
| 363 |
+
target_offer_ids = evidence_offer_ids_for_query(retrieval_query)
|
| 364 |
+
grounded = llm.extract_mentorship_course_access(
|
| 365 |
+
query,
|
| 366 |
+
selected_pages,
|
| 367 |
+
target_offer_ids=target_offer_ids,
|
| 368 |
+
)
|
| 369 |
+
if grounded is None:
|
| 370 |
+
grounded = await asyncio.to_thread(
|
| 371 |
+
llm.generate_grounded_answer,
|
| 372 |
+
prompt,
|
| 373 |
+
selected_pages,
|
| 374 |
+
query=query,
|
| 375 |
+
target_offer_ids=target_offer_ids,
|
| 376 |
+
)
|
| 377 |
+
usage = grounded.usage
|
| 378 |
+
latency_ms = grounded.latency_ms
|
| 379 |
+
if grounded.is_answered:
|
| 380 |
+
answer = grounded.answer
|
| 381 |
+
response_status = "answered"
|
| 382 |
+
sources = sources_from_pages(
|
| 383 |
+
[
|
| 384 |
+
{
|
| 385 |
+
"title": chunk.title,
|
| 386 |
+
"url": chunk.url,
|
| 387 |
+
"kind": chunk.kind,
|
| 388 |
+
}
|
| 389 |
+
for chunk in grounded.cited_chunks
|
| 390 |
+
]
|
| 391 |
+
)
|
| 392 |
+
else:
|
| 393 |
+
answer = _insufficient_evidence_answer()
|
| 394 |
+
if grounded.status == "validation_failure":
|
| 395 |
+
error_message = (
|
| 396 |
+
"grounding validation failed: "
|
| 397 |
+
f"{grounded.validation_error[:400]}"
|
| 398 |
+
)
|
| 399 |
except Exception:
|
| 400 |
error_message = "helper generation failed"
|
| 401 |
logger.exception("helper generation failed")
|
| 402 |
+
answer = _insufficient_evidence_answer()
|
| 403 |
+
sources = []
|
| 404 |
+
response_status = "insufficient_evidence"
|
| 405 |
|
| 406 |
monitor = HelperMonitor(
|
| 407 |
query=query,
|
|
|
|
| 421 |
answer=answer,
|
| 422 |
threadId=thread_id,
|
| 423 |
sources=[SourceOut(**source) for source in sources],
|
| 424 |
+
status=response_status,
|
| 425 |
usage=usage,
|
| 426 |
)
|
tai_helper/catalog.py
CHANGED
|
@@ -1,19 +1,271 @@
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
|
|
|
| 3 |
import json
|
|
|
|
| 4 |
import re
|
|
|
|
|
|
|
| 5 |
from functools import lru_cache
|
| 6 |
from typing import Any
|
| 7 |
-
from urllib.parse import urlparse
|
| 8 |
|
| 9 |
-
from .settings import repo_root
|
| 10 |
|
| 11 |
-
WORD_RE = re.compile(r"[a-z0-9]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
|
| 13 |
|
| 14 |
-
@lru_cache(maxsize=1)
|
| 15 |
def pages_payload() -> dict[str, Any]:
|
| 16 |
-
return
|
|
|
|
|
|
|
|
|
|
| 17 |
|
| 18 |
|
| 19 |
@lru_cache(maxsize=1)
|
|
@@ -25,12 +277,735 @@ def forced_prompts() -> list[str]:
|
|
| 25 |
return list(assistant_notes()["forced_prompts"])
|
| 26 |
|
| 27 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
def pages() -> list[dict[str, Any]]:
|
| 29 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
|
| 32 |
def tokenize(text: str) -> set[str]:
|
| 33 |
-
return
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 34 |
|
| 35 |
|
| 36 |
def normalized_path(url: str) -> tuple[str, str]:
|
|
@@ -42,100 +1017,483 @@ def normalized_path(url: str) -> tuple[str, str]:
|
|
| 42 |
|
| 43 |
def allowed_paths_by_host() -> dict[str, list[str]]:
|
| 44 |
result: dict[str, set[str]] = {}
|
| 45 |
-
for page in
|
| 46 |
-
host = str(
|
| 47 |
-
|
|
|
|
|
|
|
|
|
|
| 48 |
if host:
|
| 49 |
result.setdefault(host, set()).add(path)
|
| 50 |
-
if host
|
| 51 |
-
result.setdefault("www.
|
|
|
|
|
|
|
|
|
|
| 52 |
return {host: sorted(paths) for host, paths in result.items()}
|
| 53 |
|
| 54 |
|
| 55 |
def page_is_allowed(url: str) -> bool:
|
| 56 |
host, path = normalized_path(url)
|
|
|
|
| 57 |
if not host:
|
| 58 |
return False
|
| 59 |
if path.startswith(("/courses/take", "/enroll", "/order", "/checkout", "/cart")):
|
| 60 |
return False
|
| 61 |
-
if path.startswith(
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
return False
|
|
|
|
|
|
|
|
|
|
|
|
|
| 63 |
allowed = allowed_paths_by_host()
|
| 64 |
return path in allowed.get(host, [])
|
| 65 |
|
| 66 |
|
| 67 |
def source_for_url(url: str) -> dict[str, Any] | None:
|
| 68 |
-
|
| 69 |
for page in pages():
|
| 70 |
-
if str(page.get("
|
| 71 |
-
str(page.get("path", "/")).rstrip("/") or "/"
|
| 72 |
-
) == path:
|
| 73 |
return page
|
| 74 |
return None
|
| 75 |
|
| 76 |
|
| 77 |
-
def
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
|
| 81 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 82 |
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 90 |
}
|
| 91 |
-
|
|
|
|
| 92 |
|
| 93 |
-
for
|
| 94 |
-
|
|
|
|
|
|
|
| 95 |
[
|
| 96 |
-
str(
|
| 97 |
-
str(
|
| 98 |
-
|
| 99 |
-
str(page.get("meta_description", "")),
|
| 100 |
-
str(page.get("text", ""))[:8000],
|
| 101 |
]
|
| 102 |
)
|
| 103 |
-
|
| 104 |
-
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
|
| 108 |
-
|
| 109 |
-
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
if
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
|
| 116 |
-
|
| 117 |
-
|
| 118 |
-
|
| 119 |
-
|
| 120 |
-
|
| 121 |
-
if
|
| 122 |
-
|
| 123 |
-
|
| 124 |
-
|
| 125 |
-
|
| 126 |
-
|
| 127 |
-
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
|
| 133 |
-
|
| 134 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 135 |
result = []
|
| 136 |
for page in selected:
|
| 137 |
url = str(page.get("url", ""))
|
| 138 |
-
if url.startswith("internal://"):
|
| 139 |
continue
|
| 140 |
result.append(
|
| 141 |
{
|
|
@@ -154,7 +1512,7 @@ def sources_from_pages(selected: list[dict[str, Any]], limit: int = 4) -> list[d
|
|
| 154 |
|
| 155 |
|
| 156 |
def in_scope(text: str, history: list[str] | None = None) -> bool:
|
| 157 |
-
|
| 158 |
allowed_terms = {
|
| 159 |
"course",
|
| 160 |
"courses",
|
|
@@ -166,8 +1524,13 @@ def in_scope(text: str, history: list[str] | None = None) -> bool:
|
|
| 166 |
"training",
|
| 167 |
"company",
|
| 168 |
"business",
|
|
|
|
|
|
|
|
|
|
|
|
|
| 169 |
"team",
|
| 170 |
"consulting",
|
|
|
|
| 171 |
"resource",
|
| 172 |
"resources",
|
| 173 |
"youtube",
|
|
@@ -181,14 +1544,60 @@ def in_scope(text: str, history: list[str] | None = None) -> bool:
|
|
| 181 |
"agents",
|
| 182 |
"genai",
|
| 183 |
"certificate",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 184 |
"refund",
|
| 185 |
"price",
|
|
|
|
|
|
|
| 186 |
"coupon",
|
| 187 |
"discount",
|
|
|
|
|
|
|
|
|
|
| 188 |
"promo",
|
|
|
|
| 189 |
"career",
|
| 190 |
}
|
| 191 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 192 |
|
| 193 |
|
| 194 |
def coupon_intent(text: str) -> bool:
|
|
@@ -202,5 +1611,12 @@ def coupon_followup(text: str, history: list[str]) -> bool:
|
|
| 202 |
lowered.count(term) for term in ("coupon", "promo code", "discount code")
|
| 203 |
)
|
| 204 |
return prior_coupon_mentions >= 2 or any(
|
| 205 |
-
term in text.lower()
|
|
|
|
| 206 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
| 3 |
+
import hashlib
|
| 4 |
import json
|
| 5 |
+
import math
|
| 6 |
import re
|
| 7 |
+
from collections import Counter, defaultdict
|
| 8 |
+
from datetime import UTC, datetime
|
| 9 |
from functools import lru_cache
|
| 10 |
from typing import Any
|
| 11 |
+
from urllib.parse import parse_qs, urlparse
|
| 12 |
|
| 13 |
+
from .settings import repo_root, settings
|
| 14 |
|
| 15 |
+
WORD_RE = re.compile(r"[a-z0-9]+(?:\+\+|#)?")
|
| 16 |
+
SPACE_RE = re.compile(r"\s+")
|
| 17 |
+
CONTENT_HASH_RE = re.compile(r"^[0-9a-f]{64}$")
|
| 18 |
+
STOP_WORDS = {
|
| 19 |
+
"a",
|
| 20 |
+
"about",
|
| 21 |
+
"am",
|
| 22 |
+
"an",
|
| 23 |
+
"and",
|
| 24 |
+
"are",
|
| 25 |
+
"as",
|
| 26 |
+
"at",
|
| 27 |
+
"be",
|
| 28 |
+
"before",
|
| 29 |
+
"buying",
|
| 30 |
+
"by",
|
| 31 |
+
"can",
|
| 32 |
+
"committing",
|
| 33 |
+
"complete",
|
| 34 |
+
"decide",
|
| 35 |
+
"deciding",
|
| 36 |
+
"do",
|
| 37 |
+
"does",
|
| 38 |
+
"find",
|
| 39 |
+
"for",
|
| 40 |
+
"from",
|
| 41 |
+
"get",
|
| 42 |
+
"help",
|
| 43 |
+
"how",
|
| 44 |
+
"i",
|
| 45 |
+
"if",
|
| 46 |
+
"in",
|
| 47 |
+
"inside",
|
| 48 |
+
"into",
|
| 49 |
+
"is",
|
| 50 |
+
"it",
|
| 51 |
+
"looking",
|
| 52 |
+
"me",
|
| 53 |
+
"more",
|
| 54 |
+
"my",
|
| 55 |
+
"need",
|
| 56 |
+
"not",
|
| 57 |
+
"of",
|
| 58 |
+
"on",
|
| 59 |
+
"or",
|
| 60 |
+
"our",
|
| 61 |
+
"please",
|
| 62 |
+
"right",
|
| 63 |
+
"that",
|
| 64 |
+
"take",
|
| 65 |
+
"the",
|
| 66 |
+
"their",
|
| 67 |
+
"this",
|
| 68 |
+
"to",
|
| 69 |
+
"us",
|
| 70 |
+
"want",
|
| 71 |
+
"we",
|
| 72 |
+
"what",
|
| 73 |
+
"which",
|
| 74 |
+
"with",
|
| 75 |
+
"you",
|
| 76 |
+
"your",
|
| 77 |
+
}
|
| 78 |
+
EVIDENCE_STATUS = "included"
|
| 79 |
+
EVIDENCE_AUTHORITIES = {
|
| 80 |
+
"academy",
|
| 81 |
+
"academy_detail",
|
| 82 |
+
"canonical",
|
| 83 |
+
"canonical_offer",
|
| 84 |
+
"official_academy",
|
| 85 |
+
"official_site",
|
| 86 |
+
"primary",
|
| 87 |
+
"product_detail",
|
| 88 |
+
}
|
| 89 |
+
MAX_FUTURE_SKEW_SECONDS = 5 * 60
|
| 90 |
+
MAX_EVIDENCE_SPAN_CHARS = 320
|
| 91 |
+
GENERIC_RETRIEVAL_TERMS = {
|
| 92 |
+
"access",
|
| 93 |
+
"academy",
|
| 94 |
+
"ai",
|
| 95 |
+
"available",
|
| 96 |
+
"best",
|
| 97 |
+
"bot",
|
| 98 |
+
"bundle",
|
| 99 |
+
"chat",
|
| 100 |
+
"choose",
|
| 101 |
+
"choice",
|
| 102 |
+
"course",
|
| 103 |
+
"detail",
|
| 104 |
+
"fact",
|
| 105 |
+
"having",
|
| 106 |
+
"include",
|
| 107 |
+
"included",
|
| 108 |
+
"learn",
|
| 109 |
+
"learning",
|
| 110 |
+
"many",
|
| 111 |
+
"mentioned",
|
| 112 |
+
"offer",
|
| 113 |
+
"option",
|
| 114 |
+
"part",
|
| 115 |
+
"price",
|
| 116 |
+
"pricing",
|
| 117 |
+
"provide",
|
| 118 |
+
"program",
|
| 119 |
+
"resource",
|
| 120 |
+
"towards",
|
| 121 |
+
"training",
|
| 122 |
+
"true",
|
| 123 |
+
"zero",
|
| 124 |
+
"one",
|
| 125 |
+
"two",
|
| 126 |
+
"three",
|
| 127 |
+
"four",
|
| 128 |
+
"five",
|
| 129 |
+
"six",
|
| 130 |
+
"seven",
|
| 131 |
+
"eight",
|
| 132 |
+
"nine",
|
| 133 |
+
"ten",
|
| 134 |
+
}
|
| 135 |
+
|
| 136 |
+
# Explicit offer names are a hard retrieval boundary, not merely a ranking hint.
|
| 137 |
+
# A truthful sentence from another course is still an unsafe answer for the
|
| 138 |
+
# course the visitor named. Preview aliases intentionally resolve to a distinct
|
| 139 |
+
# offer so paid-course entitlements cannot leak into preview answers.
|
| 140 |
+
OFFER_QUERY_ALIASES: dict[str, tuple[str, ...]] = {
|
| 141 |
+
"full-stack-ai-engineering": (
|
| 142 |
+
"full stack ai engineering",
|
| 143 |
+
"full stack engineering",
|
| 144 |
+
"full stack",
|
| 145 |
+
),
|
| 146 |
+
"agent-engineering": (
|
| 147 |
+
"agentic ai engineering",
|
| 148 |
+
"agent engineering",
|
| 149 |
+
),
|
| 150 |
+
"llm-primer": (
|
| 151 |
+
"10-hour llm fundamentals",
|
| 152 |
+
"10 hour llm fundamentals",
|
| 153 |
+
"llm fundamentals",
|
| 154 |
+
"llm primer",
|
| 155 |
+
),
|
| 156 |
+
"python-for-ai-engineering": (
|
| 157 |
+
"beginner python for ai engineering",
|
| 158 |
+
"python for ai engineering",
|
| 159 |
+
"beginner python",
|
| 160 |
+
"python course",
|
| 161 |
+
),
|
| 162 |
+
"ai-for-work": (
|
| 163 |
+
"master ai for work",
|
| 164 |
+
"ai for work",
|
| 165 |
+
),
|
| 166 |
+
"building-llms-for-production": (
|
| 167 |
+
"building llms for production",
|
| 168 |
+
"building llm for production",
|
| 169 |
+
"the ebook",
|
| 170 |
+
"e-book",
|
| 171 |
+
"ebook",
|
| 172 |
+
),
|
| 173 |
+
"get-it-all": (
|
| 174 |
+
"get it all bundle",
|
| 175 |
+
"get-it-all bundle",
|
| 176 |
+
"get it all",
|
| 177 |
+
),
|
| 178 |
+
"from-coding-novice-to-advanced-llm-developer": (
|
| 179 |
+
"from non-coder to ai engineer",
|
| 180 |
+
"from non coder to ai engineer",
|
| 181 |
+
"non-coder to ai engineer bundle",
|
| 182 |
+
),
|
| 183 |
+
"10-hour-crash-course-into-llm-developer-expert": (
|
| 184 |
+
"from developer to advanced ai engineer",
|
| 185 |
+
"developer to advanced ai engineer bundle",
|
| 186 |
+
),
|
| 187 |
+
"mentorship": (
|
| 188 |
+
"towards ai mentorship",
|
| 189 |
+
"mentorship",
|
| 190 |
+
"membership",
|
| 191 |
+
),
|
| 192 |
+
}
|
| 193 |
+
|
| 194 |
+
PREVIEW_OFFER_IDS = {
|
| 195 |
+
"full-stack-ai-engineering": "full-stack-ai-engineering-free-preview",
|
| 196 |
+
"agent-engineering": "agent-engineering-free-preview",
|
| 197 |
+
}
|
| 198 |
+
|
| 199 |
+
# The helper was historically embedded on these exact towardsai.net pages. Keep
|
| 200 |
+
# them available for widget routing, but never add them to the evidence corpus.
|
| 201 |
+
LEGACY_PUBLIC_PATHS_BY_HOST: dict[str, frozenset[str]] = {
|
| 202 |
+
"towardsai.net": frozenset(
|
| 203 |
+
{
|
| 204 |
+
"/",
|
| 205 |
+
"/academy",
|
| 206 |
+
"/b2b",
|
| 207 |
+
"/book",
|
| 208 |
+
"/community",
|
| 209 |
+
"/let-us-transform-your-team-into-ai-first-employees-to-stay-ahead-of-competitors",
|
| 210 |
+
"/towards-ai-resource-library",
|
| 211 |
+
}
|
| 212 |
+
),
|
| 213 |
+
}
|
| 214 |
+
|
| 215 |
+
|
| 216 |
+
def _catalog_signature(path: str) -> tuple[str, int, int]:
|
| 217 |
+
catalog_path = repo_root() / "data" / path
|
| 218 |
+
try:
|
| 219 |
+
stat = catalog_path.stat()
|
| 220 |
+
except OSError:
|
| 221 |
+
return path, 0, 0
|
| 222 |
+
return path, stat.st_mtime_ns, stat.st_size
|
| 223 |
+
|
| 224 |
+
|
| 225 |
+
def _load_catalog(
|
| 226 |
+
path: str, _signature: tuple[str, int, int]
|
| 227 |
+
) -> tuple[list[dict[str, Any]], str]:
|
| 228 |
+
catalog_path = repo_root() / "data" / path
|
| 229 |
+
if not catalog_path.is_file():
|
| 230 |
+
return [], ""
|
| 231 |
+
try:
|
| 232 |
+
payload = json.loads(catalog_path.read_text())
|
| 233 |
+
except (OSError, UnicodeError, json.JSONDecodeError):
|
| 234 |
+
return [], ""
|
| 235 |
+
if not isinstance(payload, dict) or not isinstance(payload.get("pages"), list):
|
| 236 |
+
return [], ""
|
| 237 |
+
generated_at = str(payload.get("generated_at", ""))
|
| 238 |
+
result = []
|
| 239 |
+
for raw_page in payload.get("pages", []):
|
| 240 |
+
page = dict(raw_page)
|
| 241 |
+
page.setdefault("catalog_generated_at", generated_at)
|
| 242 |
+
result.append(page)
|
| 243 |
+
return result, generated_at
|
| 244 |
+
|
| 245 |
+
|
| 246 |
+
@lru_cache(maxsize=8)
|
| 247 |
+
def _pages_payload_cached(
|
| 248 |
+
academy_signature: tuple[str, int, int],
|
| 249 |
+
website_signature: tuple[str, int, int],
|
| 250 |
+
) -> dict[str, Any]:
|
| 251 |
+
academy_pages, academy_generated_at = _load_catalog("pages.json", academy_signature)
|
| 252 |
+
website_pages, website_generated_at = _load_catalog(
|
| 253 |
+
"towardsai_com_pages.json", website_signature
|
| 254 |
+
)
|
| 255 |
+
return {
|
| 256 |
+
"pages": [*academy_pages, *website_pages],
|
| 257 |
+
"generated_at": {
|
| 258 |
+
"academy": academy_generated_at,
|
| 259 |
+
"website": website_generated_at,
|
| 260 |
+
},
|
| 261 |
+
}
|
| 262 |
|
| 263 |
|
|
|
|
| 264 |
def pages_payload() -> dict[str, Any]:
|
| 265 |
+
return _pages_payload_cached(
|
| 266 |
+
_catalog_signature("pages.json"),
|
| 267 |
+
_catalog_signature("towardsai_com_pages.json"),
|
| 268 |
+
)
|
| 269 |
|
| 270 |
|
| 271 |
@lru_cache(maxsize=1)
|
|
|
|
| 277 |
return list(assistant_notes()["forced_prompts"])
|
| 278 |
|
| 279 |
|
| 280 |
+
def _canonical_key(url: str) -> tuple[str, str]:
|
| 281 |
+
host, path = normalized_path(url)
|
| 282 |
+
host = host.removeprefix("www.")
|
| 283 |
+
return host, path
|
| 284 |
+
|
| 285 |
+
|
| 286 |
+
def _authority(page: dict[str, Any]) -> float:
|
| 287 |
+
raw = page.get("authority", "")
|
| 288 |
+
if isinstance(raw, (int, float)):
|
| 289 |
+
return float(raw)
|
| 290 |
+
labels = {
|
| 291 |
+
"canonical": 5.0,
|
| 292 |
+
"canonical_offer": 5.0,
|
| 293 |
+
"official_site": 5.0,
|
| 294 |
+
"primary": 5.0,
|
| 295 |
+
"academy_detail": 4.0,
|
| 296 |
+
"official_academy": 4.0,
|
| 297 |
+
"product_detail": 4.0,
|
| 298 |
+
"academy": 3.5,
|
| 299 |
+
"catalog": 3.0,
|
| 300 |
+
"external": 2.0,
|
| 301 |
+
"curated_external": 2.0,
|
| 302 |
+
"legacy": 1.0,
|
| 303 |
+
}
|
| 304 |
+
if str(raw).lower() in labels:
|
| 305 |
+
return labels[str(raw).lower()]
|
| 306 |
+
host = str(page.get("host") or urlparse(str(page.get("url", ""))).hostname or "")
|
| 307 |
+
if host.removeprefix("www.") == "towardsai.com":
|
| 308 |
+
return 5.0
|
| 309 |
+
if host == "academy.towardsai.net":
|
| 310 |
+
return 3.5
|
| 311 |
+
return 2.0
|
| 312 |
+
|
| 313 |
+
|
| 314 |
+
def _parse_timestamp(raw: Any) -> datetime | None:
|
| 315 |
+
if not raw:
|
| 316 |
+
return None
|
| 317 |
+
try:
|
| 318 |
+
value = str(raw).replace("Z", "+00:00")
|
| 319 |
+
parsed = datetime.fromisoformat(value)
|
| 320 |
+
if parsed.tzinfo is None:
|
| 321 |
+
parsed = parsed.replace(tzinfo=UTC)
|
| 322 |
+
return parsed.astimezone(UTC)
|
| 323 |
+
except (TypeError, ValueError):
|
| 324 |
+
return None
|
| 325 |
+
|
| 326 |
+
|
| 327 |
+
def _is_fresh(page: dict[str, Any]) -> bool:
|
| 328 |
+
# A successful per-page fetch time is required. The catalog generation time
|
| 329 |
+
# cannot make reused or failed content appear current.
|
| 330 |
+
fetched_at = _parse_timestamp(page.get("fetched_at"))
|
| 331 |
+
if fetched_at is None:
|
| 332 |
+
return False
|
| 333 |
+
age_seconds = (datetime.now(UTC) - fetched_at).total_seconds()
|
| 334 |
+
max_age_seconds = max(settings.catalog_max_age_days, 0) * 24 * 60 * 60
|
| 335 |
+
return -MAX_FUTURE_SKEW_SECONDS <= age_seconds <= max_age_seconds
|
| 336 |
+
|
| 337 |
+
|
| 338 |
+
def _is_included(page: dict[str, Any]) -> bool:
|
| 339 |
+
return str(page.get("status", "")).lower() == EVIDENCE_STATUS and not bool(
|
| 340 |
+
page.get("excluded")
|
| 341 |
+
)
|
| 342 |
+
|
| 343 |
+
|
| 344 |
+
def _is_evidence_page(page: dict[str, Any]) -> bool:
|
| 345 |
+
raw_chunks = page.get("chunks")
|
| 346 |
+
text = str(page.get("text", ""))
|
| 347 |
+
normalized_page_text = SPACE_RE.sub(" ", text).strip()
|
| 348 |
+
page_url_text = str(page.get("url", ""))
|
| 349 |
+
canonical_url = str(page.get("canonical_url") or page.get("url") or "")
|
| 350 |
+
page_url = urlparse(page_url_text)
|
| 351 |
+
canonical = urlparse(canonical_url)
|
| 352 |
+
valid_chunks = (
|
| 353 |
+
isinstance(raw_chunks, list)
|
| 354 |
+
and bool(raw_chunks)
|
| 355 |
+
and all(
|
| 356 |
+
isinstance(chunk, dict)
|
| 357 |
+
and bool(str(chunk.get("chunk_id", "")).strip())
|
| 358 |
+
and bool(str(chunk.get("text", "")).strip())
|
| 359 |
+
and isinstance(chunk.get("evidence_spans"), list)
|
| 360 |
+
and bool(chunk["evidence_spans"])
|
| 361 |
+
and all(
|
| 362 |
+
isinstance(span, dict)
|
| 363 |
+
and bool(str(span.get("span_id", "")).strip())
|
| 364 |
+
and bool(str(span.get("text", "")).strip())
|
| 365 |
+
and len(SPACE_RE.sub(" ", str(span["text"])).strip())
|
| 366 |
+
<= MAX_EVIDENCE_SPAN_CHARS
|
| 367 |
+
and SPACE_RE.sub(" ", str(span["text"])).strip()
|
| 368 |
+
in SPACE_RE.sub(" ", str(chunk["text"])).strip()
|
| 369 |
+
and SPACE_RE.sub(" ", str(span["text"])).strip() in normalized_page_text
|
| 370 |
+
for span in chunk["evidence_spans"]
|
| 371 |
+
)
|
| 372 |
+
and len({str(span["span_id"]).strip() for span in chunk["evidence_spans"]})
|
| 373 |
+
== len(chunk["evidence_spans"])
|
| 374 |
+
for chunk in raw_chunks
|
| 375 |
+
)
|
| 376 |
+
and len({str(chunk["chunk_id"]).strip() for chunk in raw_chunks})
|
| 377 |
+
== len(raw_chunks)
|
| 378 |
+
)
|
| 379 |
+
content_hash = str(page.get("content_hash", ""))
|
| 380 |
+
evidence_hash = str(page.get("evidence_hash", ""))
|
| 381 |
+
canonical_evidence = json.dumps(
|
| 382 |
+
{
|
| 383 |
+
key: value
|
| 384 |
+
for key, value in page.items()
|
| 385 |
+
if key not in {"catalog_generated_at", "evidence_hash"}
|
| 386 |
+
},
|
| 387 |
+
ensure_ascii=False,
|
| 388 |
+
sort_keys=True,
|
| 389 |
+
separators=(",", ":"),
|
| 390 |
+
)
|
| 391 |
+
return (
|
| 392 |
+
str(page.get("status", "")).lower() == EVIDENCE_STATUS
|
| 393 |
+
and page.get("retrieval_eligible") is True
|
| 394 |
+
and str(page.get("authority", "")).lower() in EVIDENCE_AUTHORITIES
|
| 395 |
+
and page.get("http_status") == 200
|
| 396 |
+
and page_url.scheme == "https"
|
| 397 |
+
and page_url.hostname in {"towardsai.com", "academy.towardsai.net"}
|
| 398 |
+
and canonical.scheme == "https"
|
| 399 |
+
and canonical.hostname in {"towardsai.com", "academy.towardsai.net"}
|
| 400 |
+
and _canonical_key(page_url_text) == _canonical_key(canonical_url)
|
| 401 |
+
and str(page.get("host", "")).lower() == (page_url.hostname or "").lower()
|
| 402 |
+
and (str(page.get("path", "")).rstrip("/") or "/")
|
| 403 |
+
== (page_url.path.rstrip("/") or "/")
|
| 404 |
+
and CONTENT_HASH_RE.fullmatch(content_hash) is not None
|
| 405 |
+
and hashlib.sha256(text.encode("utf-8")).hexdigest() == content_hash
|
| 406 |
+
and CONTENT_HASH_RE.fullmatch(evidence_hash) is not None
|
| 407 |
+
and hashlib.sha256(canonical_evidence.encode("utf-8")).hexdigest()
|
| 408 |
+
== evidence_hash
|
| 409 |
+
and valid_chunks
|
| 410 |
+
and _is_fresh(page)
|
| 411 |
+
)
|
| 412 |
+
|
| 413 |
+
|
| 414 |
+
def all_pages() -> list[dict[str, Any]]:
|
| 415 |
+
"""Return known safe pages, including stale pages used only for widget routing."""
|
| 416 |
+
return [page for page in pages_payload()["pages"] if _is_included(page)]
|
| 417 |
+
|
| 418 |
+
|
| 419 |
+
def _page_preference(page: dict[str, Any]) -> tuple[float, float, int]:
|
| 420 |
+
fetched_at = _parse_timestamp(page.get("fetched_at"))
|
| 421 |
+
fetched_value = fetched_at.timestamp() if fetched_at else 0.0
|
| 422 |
+
return _authority(page), fetched_value, len(str(page.get("text", "")))
|
| 423 |
+
|
| 424 |
+
|
| 425 |
+
def _fresh_pages() -> tuple[dict[str, Any], ...]:
|
| 426 |
+
best_by_url: dict[tuple[str, str], dict[str, Any]] = {}
|
| 427 |
+
for page in all_pages():
|
| 428 |
+
if not _is_evidence_page(page):
|
| 429 |
+
continue
|
| 430 |
+
key = _canonical_key(str(page.get("url", "")))
|
| 431 |
+
if not all(key):
|
| 432 |
+
continue
|
| 433 |
+
previous = best_by_url.get(key)
|
| 434 |
+
if previous is None or _page_preference(page) > _page_preference(previous):
|
| 435 |
+
best_by_url[key] = page
|
| 436 |
+
|
| 437 |
+
# A canonical offer page wins over lower-authority mirrors for the same
|
| 438 |
+
# product. This prevents conflicting catalog/Thinkific figures from both
|
| 439 |
+
# entering one evidence set.
|
| 440 |
+
best_by_offer: dict[str, dict[str, Any]] = {}
|
| 441 |
+
pages_without_offer: list[dict[str, Any]] = []
|
| 442 |
+
for page in best_by_url.values():
|
| 443 |
+
offer_id = str(page.get("offer_id") or "").strip()
|
| 444 |
+
if not offer_id:
|
| 445 |
+
pages_without_offer.append(page)
|
| 446 |
+
continue
|
| 447 |
+
previous = best_by_offer.get(offer_id)
|
| 448 |
+
if previous is None or _page_preference(page) > _page_preference(previous):
|
| 449 |
+
best_by_offer[offer_id] = page
|
| 450 |
+
candidates = (*pages_without_offer, *best_by_offer.values())
|
| 451 |
+
chunk_id_counts = Counter(
|
| 452 |
+
str(chunk["chunk_id"]).strip()
|
| 453 |
+
for page in candidates
|
| 454 |
+
for chunk in page["chunks"]
|
| 455 |
+
)
|
| 456 |
+
return tuple(
|
| 457 |
+
page
|
| 458 |
+
for page in candidates
|
| 459 |
+
if all(
|
| 460 |
+
chunk_id_counts[str(chunk["chunk_id"]).strip()] == 1
|
| 461 |
+
for chunk in page["chunks"]
|
| 462 |
+
)
|
| 463 |
+
)
|
| 464 |
+
|
| 465 |
+
|
| 466 |
def pages() -> list[dict[str, Any]]:
|
| 467 |
+
"""Return only fresh, reviewed pages that may be used as factual evidence."""
|
| 468 |
+
return list(_fresh_pages())
|
| 469 |
+
|
| 470 |
+
|
| 471 |
+
def _token_list(text: str) -> list[str]:
|
| 472 |
+
tokens = []
|
| 473 |
+
for raw_token in WORD_RE.findall(text.lower()):
|
| 474 |
+
token = raw_token.lower()
|
| 475 |
+
if len(token) <= 1 or token in STOP_WORDS:
|
| 476 |
+
continue
|
| 477 |
+
if token == "classes":
|
| 478 |
+
token = "class"
|
| 479 |
+
elif token.endswith("ies") and len(token) > 4:
|
| 480 |
+
token = f"{token[:-3]}y"
|
| 481 |
+
elif token.endswith("s") and len(token) > 3 and not token.endswith("ss"):
|
| 482 |
+
token = token[:-1]
|
| 483 |
+
tokens.append(token)
|
| 484 |
+
return tokens
|
| 485 |
|
| 486 |
|
| 487 |
def tokenize(text: str) -> set[str]:
|
| 488 |
+
return set(_token_list(text))
|
| 489 |
+
|
| 490 |
+
|
| 491 |
+
def offer_ids_for_query(query: str) -> frozenset[str]:
|
| 492 |
+
"""Resolve explicitly named offers into a fail-closed retrieval boundary."""
|
| 493 |
+
|
| 494 |
+
lowered = SPACE_RE.sub(" ", query.casefold().replace("–", "-")).strip()
|
| 495 |
+
if "webinar" in lowered:
|
| 496 |
+
return frozenset()
|
| 497 |
+
matches = {
|
| 498 |
+
offer_id
|
| 499 |
+
for offer_id, aliases in OFFER_QUERY_ALIASES.items()
|
| 500 |
+
if any(alias in lowered for alias in aliases)
|
| 501 |
+
}
|
| 502 |
+
preview_intent = bool(
|
| 503 |
+
re.search(r"\bpreview\b|\bfree\s+lessons?\b", lowered)
|
| 504 |
+
)
|
| 505 |
+
if preview_intent:
|
| 506 |
+
for paid_offer_id, preview_offer_id in PREVIEW_OFFER_IDS.items():
|
| 507 |
+
if paid_offer_id in matches:
|
| 508 |
+
matches.remove(paid_offer_id)
|
| 509 |
+
matches.add(preview_offer_id)
|
| 510 |
+
|
| 511 |
+
if "mentorship" in matches and len(matches) > 1:
|
| 512 |
+
compares_offers = bool(
|
| 513 |
+
re.search(r"\b(?:compare|versus|vs\.?|between)\b", lowered)
|
| 514 |
+
)
|
| 515 |
+
course_subject_includes_mentorship = bool(
|
| 516 |
+
re.search(
|
| 517 |
+
r"(?:full stack|agent(?:ic)? engineering|llm fundamentals|llm primer|"
|
| 518 |
+
r"python for ai|ai for work|building llms|get it all).{0,80}"
|
| 519 |
+
r"\b(?:include|come with|provide|offer|have)\b.{0,50}\bmentor",
|
| 520 |
+
lowered,
|
| 521 |
+
)
|
| 522 |
+
)
|
| 523 |
+
if course_subject_includes_mentorship:
|
| 524 |
+
matches.remove("mentorship")
|
| 525 |
+
elif not compares_offers:
|
| 526 |
+
# When a course is mentioned inside a mentorship access/discount/
|
| 527 |
+
# cancellation question, the mentorship page is the authority.
|
| 528 |
+
matches = {"mentorship"}
|
| 529 |
+
return frozenset(matches)
|
| 530 |
+
|
| 531 |
+
|
| 532 |
+
def evidence_offer_ids_for_query(query: str) -> frozenset[str]:
|
| 533 |
+
"""Return offer IDs allowed to supply evidence for this exact question."""
|
| 534 |
+
|
| 535 |
+
targets = set(offer_ids_for_query(query))
|
| 536 |
+
fields = requested_fact_fields(query)
|
| 537 |
+
if not fields:
|
| 538 |
+
return frozenset(targets)
|
| 539 |
+
evidence_targets: set[str] = set()
|
| 540 |
+
for field in fields:
|
| 541 |
+
evidence_targets.update(evidence_offer_ids_for_field(query, field))
|
| 542 |
+
return frozenset(evidence_targets)
|
| 543 |
+
|
| 544 |
+
|
| 545 |
+
def evidence_offer_ids_for_field(query: str, field: str) -> frozenset[str]:
|
| 546 |
+
"""Return the offers allowed to prove one requested fact field.
|
| 547 |
+
|
| 548 |
+
Evidence exceptions are deliberately field-specific. A paid course page can
|
| 549 |
+
state the size of its free preview, but that exception must never expose the
|
| 550 |
+
paid course's access entitlement to a preview access question in the same
|
| 551 |
+
visitor message.
|
| 552 |
+
"""
|
| 553 |
+
|
| 554 |
+
targets = set(offer_ids_for_query(query))
|
| 555 |
+
# A paid offer page may explicitly state how many lessons its free preview
|
| 556 |
+
# contains. This is the sole preview↔parent exception; paid entitlements,
|
| 557 |
+
# certificates, prices, and refund terms remain outside the preview boundary.
|
| 558 |
+
if field == "lesson_count":
|
| 559 |
+
reverse_preview_ids = {
|
| 560 |
+
preview_id: paid_id for paid_id, preview_id in PREVIEW_OFFER_IDS.items()
|
| 561 |
+
}
|
| 562 |
+
targets.update(
|
| 563 |
+
reverse_preview_ids[offer_id]
|
| 564 |
+
for offer_id in tuple(targets)
|
| 565 |
+
if offer_id in reverse_preview_ids
|
| 566 |
+
)
|
| 567 |
+
return frozenset(targets)
|
| 568 |
+
|
| 569 |
+
|
| 570 |
+
def offer_alias_tokens(offer_ids: frozenset[str] | set[str]) -> frozenset[str]:
|
| 571 |
+
"""Return tokens that name the selected offers, for query relevance checks."""
|
| 572 |
+
|
| 573 |
+
result: set[str] = set()
|
| 574 |
+
reverse_preview_ids = {
|
| 575 |
+
preview_id: paid_id for paid_id, preview_id in PREVIEW_OFFER_IDS.items()
|
| 576 |
+
}
|
| 577 |
+
for offer_id in offer_ids:
|
| 578 |
+
paid_offer_id = reverse_preview_ids.get(offer_id, offer_id)
|
| 579 |
+
for alias in OFFER_QUERY_ALIASES.get(paid_offer_id, ()):
|
| 580 |
+
result.update(_token_list(alias))
|
| 581 |
+
# Query validation tokenizes without stemming, so retain the exact
|
| 582 |
+
# alias words as well as their retrieval-normalized forms.
|
| 583 |
+
result.update(WORD_RE.findall(alias.casefold()))
|
| 584 |
+
if offer_id in reverse_preview_ids:
|
| 585 |
+
result.update({"free", "preview", "lesson", "lessons"})
|
| 586 |
+
return frozenset(result)
|
| 587 |
+
|
| 588 |
+
|
| 589 |
+
BUNDLE_OFFER_IDS = frozenset(
|
| 590 |
+
{
|
| 591 |
+
"get-it-all",
|
| 592 |
+
"from-coding-novice-to-advanced-llm-developer",
|
| 593 |
+
"10-hour-crash-course-into-llm-developer-expert",
|
| 594 |
+
}
|
| 595 |
+
)
|
| 596 |
+
|
| 597 |
+
TOTAL_LESSON_PATTERNS: dict[str, re.Pattern[str]] = {
|
| 598 |
+
"full-stack-ai-engineering": re.compile(r"\b92\s+lessons\b", re.IGNORECASE),
|
| 599 |
+
"agent-engineering": re.compile(r"\b35\s+lessons\b", re.IGNORECASE),
|
| 600 |
+
"python-for-ai-engineering": re.compile(r"\b38\s+lessons\b", re.IGNORECASE),
|
| 601 |
+
"ai-for-work": re.compile(r"\b98\s+lessons\b", re.IGNORECASE),
|
| 602 |
+
"building-llms-for-production": re.compile(
|
| 603 |
+
r"\b84\s+lessons\b", re.IGNORECASE
|
| 604 |
+
),
|
| 605 |
+
"agent-engineering-free-preview": re.compile(
|
| 606 |
+
r"\b7\s+(?:free\s+|full\s+)?lessons\b", re.IGNORECASE
|
| 607 |
+
),
|
| 608 |
+
"full-stack-ai-engineering-free-preview": re.compile(
|
| 609 |
+
r"\b(?:first\s+)?6\s+(?:free\s+|preview\s+)?lessons\b", re.IGNORECASE
|
| 610 |
+
),
|
| 611 |
+
}
|
| 612 |
+
|
| 613 |
+
DURATION_PATTERNS: dict[str, re.Pattern[str]] = {
|
| 614 |
+
"full-stack-ai-engineering": re.compile(
|
| 615 |
+
r"\b60\+\s*hours\b", re.IGNORECASE
|
| 616 |
+
),
|
| 617 |
+
"llm-primer": re.compile(
|
| 618 |
+
r"\b10\s*hours?\b|\bfive\s+(?:in-depth\s+)?2-hour\s+(?:video\s+)?sessions\b",
|
| 619 |
+
re.IGNORECASE,
|
| 620 |
+
),
|
| 621 |
+
"ai-for-work": re.compile(r"\baround\s+20\s+hours\b", re.IGNORECASE),
|
| 622 |
+
}
|
| 623 |
+
|
| 624 |
+
MONTHLY_PLAN_PRICE_RE = re.compile(
|
| 625 |
+
r"\$\s*99(?:\.00)?\s*(?:/\s*month|monthly)\b",
|
| 626 |
+
re.IGNORECASE,
|
| 627 |
+
)
|
| 628 |
+
YEARLY_PLAN_PRICE_RE = re.compile(
|
| 629 |
+
r"\$\s*899(?:\.00)?.{0,20}(?:/\s*year|per year|once a year)\b",
|
| 630 |
+
re.IGNORECASE,
|
| 631 |
+
)
|
| 632 |
+
PERMANENT_ACCESS_RE = re.compile(
|
| 633 |
+
r"\b(?:lifetime|forever|keep|retain|retained)\b",
|
| 634 |
+
re.IGNORECASE,
|
| 635 |
+
)
|
| 636 |
+
|
| 637 |
+
|
| 638 |
+
def monthly_plan_intent(query: str) -> bool:
|
| 639 |
+
return bool(
|
| 640 |
+
re.search(
|
| 641 |
+
r"\bmonthly\b|\bmonth[- ](?:to|by)[- ]month\b|"
|
| 642 |
+
r"\b(?:per|each)\s+month\b|"
|
| 643 |
+
r"\bmonth\s+(?:price|cost|plan|rate|subscription|membership|mentorship)\b",
|
| 644 |
+
query,
|
| 645 |
+
)
|
| 646 |
+
)
|
| 647 |
+
|
| 648 |
+
|
| 649 |
+
def yearly_plan_intent(query: str) -> bool:
|
| 650 |
+
return bool(
|
| 651 |
+
re.search(
|
| 652 |
+
r"\b(?:yearly|annual|annually)\b|\bper\s+year\b|"
|
| 653 |
+
r"\byear\s+(?:price|cost|plan|rate|subscription|membership|mentorship)\b",
|
| 654 |
+
query,
|
| 655 |
+
)
|
| 656 |
+
)
|
| 657 |
+
|
| 658 |
+
|
| 659 |
+
def _supports_monthly_plan_price(text: str) -> bool:
|
| 660 |
+
for match in MONTHLY_PLAN_PRICE_RE.finditer(text):
|
| 661 |
+
prefix = text[max(0, match.start() - 16) : match.start()]
|
| 662 |
+
suffix = text[match.end() : match.end() + 32]
|
| 663 |
+
if re.search(r"\b(?:from|about)\s*$", prefix):
|
| 664 |
+
continue
|
| 665 |
+
if re.search(r"\bbilled\s+(?:yearly|annually)\b", suffix):
|
| 666 |
+
continue
|
| 667 |
+
return True
|
| 668 |
+
return False
|
| 669 |
+
|
| 670 |
+
|
| 671 |
+
def _supports_yearly_plan_price(text: str) -> bool:
|
| 672 |
+
for match in YEARLY_PLAN_PRICE_RE.finditer(text):
|
| 673 |
+
prefix = text[max(0, match.start() - 16) : match.start()]
|
| 674 |
+
if re.search(r"\b(?:from|about)\s*$", prefix):
|
| 675 |
+
continue
|
| 676 |
+
if "a month" in match.group().casefold():
|
| 677 |
+
continue
|
| 678 |
+
return True
|
| 679 |
+
return False
|
| 680 |
+
|
| 681 |
+
|
| 682 |
+
def requested_fact_fields(query: str) -> frozenset[str]:
|
| 683 |
+
"""Classify high-risk facts that require typed, target-bound evidence."""
|
| 684 |
+
|
| 685 |
+
lowered = SPACE_RE.sub(" ", query.casefold().replace("–", "-")).strip()
|
| 686 |
+
fields: set[str] = set()
|
| 687 |
+
if re.search(r"\b(?:price|cost|how much)\b|[$€£¥]\s*\d", lowered):
|
| 688 |
+
fields.add("price")
|
| 689 |
+
if "lesson" in lowered and re.search(
|
| 690 |
+
r"\b(?:how many|number of|total|overall)\b|\b\d+\s+lessons?\b|\blessons?\s*[:=]?\s*\d+\b",
|
| 691 |
+
lowered,
|
| 692 |
+
):
|
| 693 |
+
fields.add("lesson_count")
|
| 694 |
+
if "product" in lowered and re.search(
|
| 695 |
+
r"\b(?:how many|number of|total)\b", lowered
|
| 696 |
+
):
|
| 697 |
+
fields.add("product_count")
|
| 698 |
+
if "page" in lowered and re.search(
|
| 699 |
+
r"\b(?:how many|number of|total|pages?)\b", lowered
|
| 700 |
+
):
|
| 701 |
+
fields.add("page_count")
|
| 702 |
+
if re.search(
|
| 703 |
+
r"\bhow long\b|\bhow many hours?\b|\bduration\b|\btime to (?:finish|complete)\b|\b\d+\+?\s*hours?\b|\bhours?\s+(?:long|total)\b|\bhours?.{0,24}\b(?:take|finish|complete)\b",
|
| 704 |
+
lowered,
|
| 705 |
+
):
|
| 706 |
+
fields.add("duration")
|
| 707 |
+
if re.search(
|
| 708 |
+
r"\bprerequisites?\b|\brequirements?\b|\bprior experience\b|\bneed to know\b|\bneed (?:python|coding|code)\b",
|
| 709 |
+
lowered,
|
| 710 |
+
):
|
| 711 |
+
fields.add("prerequisite")
|
| 712 |
+
if re.search(r"\bcertificat(?:e|ion|ed)\b", lowered):
|
| 713 |
+
fields.add("certificate")
|
| 714 |
+
if re.search(r"\brefund\b|\bmoney[- ]back\b", lowered):
|
| 715 |
+
fields.add("refund")
|
| 716 |
+
if re.search(r"\bguarantee(?:d)?\b", lowered):
|
| 717 |
+
fields.add("guarantee")
|
| 718 |
+
if re.search(
|
| 719 |
+
r"\bdiscount\b|\bcoupon\b|\bpromo\b|\bsavings?\b|"
|
| 720 |
+
r"\b\d+(?:\.\d+)?%\s+(?:off|less)\b|"
|
| 721 |
+
r"\bsaves?\s+\d+(?:\.\d+)?%",
|
| 722 |
+
lowered,
|
| 723 |
+
):
|
| 724 |
+
fields.add("discount")
|
| 725 |
+
if re.search(r"\baccess\b|\blifetime\b|\bkeep\b|\bforever\b|\bretain\b", lowered):
|
| 726 |
+
fields.add("access")
|
| 727 |
+
if (
|
| 728 |
+
"mentor" in lowered
|
| 729 |
+
and not re.search(r"\bcancel|\bcancell|\bafter\b|\blifetime\b|\bkeep\b", lowered)
|
| 730 |
+
and re.search(
|
| 731 |
+
r"\b(?:course|courses|llm fundamentals|full stack|agent engineering|master ai for work|choice|choose)\b",
|
| 732 |
+
lowered,
|
| 733 |
+
)
|
| 734 |
+
and re.search(r"\b(?:included|include|access|choice|choose)\b", lowered)
|
| 735 |
+
):
|
| 736 |
+
fields.add("inclusion")
|
| 737 |
+
return frozenset(fields)
|
| 738 |
+
|
| 739 |
+
|
| 740 |
+
def mixed_preview_fact_request(query: str) -> bool:
|
| 741 |
+
"""Return true when one message mixes multiple preview/paid fact scopes.
|
| 742 |
+
|
| 743 |
+
A single offer ID cannot safely assign separate clauses to the free preview
|
| 744 |
+
and paid course. These compound questions therefore fail closed and ask the
|
| 745 |
+
visitor to use the contact form (or ask the facts separately).
|
| 746 |
+
"""
|
| 747 |
+
|
| 748 |
+
lowered = SPACE_RE.sub(" ", query.casefold()).strip()
|
| 749 |
+
preview_intent = bool(re.search(r"\bpreview\b|\bfree\s+lessons?\b", lowered))
|
| 750 |
+
return preview_intent and len(requested_fact_fields(query)) > 1
|
| 751 |
+
|
| 752 |
+
|
| 753 |
+
def _span_supports_fact(
|
| 754 |
+
span_text: str,
|
| 755 |
+
*,
|
| 756 |
+
field: str,
|
| 757 |
+
offer_id: str,
|
| 758 |
+
query: str,
|
| 759 |
+
) -> bool:
|
| 760 |
+
text = SPACE_RE.sub(" ", span_text.casefold().replace("–", "-")).strip()
|
| 761 |
+
lowered_query = query.casefold().replace("–", "-")
|
| 762 |
+
is_preview = offer_id.endswith("-free-preview")
|
| 763 |
+
|
| 764 |
+
if field == "price":
|
| 765 |
+
if offer_id == "get-it-all":
|
| 766 |
+
asks_where_price_is_displayed = bool(
|
| 767 |
+
"checkout" in lowered_query
|
| 768 |
+
and re.search(
|
| 769 |
+
r"\b(?:where|shown|displayed|find|see)\b", lowered_query
|
| 770 |
+
)
|
| 771 |
+
)
|
| 772 |
+
return (
|
| 773 |
+
asks_where_price_is_displayed
|
| 774 |
+
and "bundle price shown at checkout" in text
|
| 775 |
+
)
|
| 776 |
+
if offer_id in BUNDLE_OFFER_IDS:
|
| 777 |
+
return "bundle price" in text or (
|
| 778 |
+
"one-time" in text and bool(_CURRENCY_TOKEN_RE.search(text))
|
| 779 |
+
)
|
| 780 |
+
if offer_id == "mentorship":
|
| 781 |
+
if yearly_plan_intent(lowered_query):
|
| 782 |
+
return _supports_yearly_plan_price(text)
|
| 783 |
+
if monthly_plan_intent(lowered_query):
|
| 784 |
+
return _supports_monthly_plan_price(text)
|
| 785 |
+
return _supports_monthly_plan_price(
|
| 786 |
+
text
|
| 787 |
+
) or _supports_yearly_plan_price(text)
|
| 788 |
+
if is_preview:
|
| 789 |
+
return "free" in text
|
| 790 |
+
return bool(_CURRENCY_TOKEN_RE.search(text)) and any(
|
| 791 |
+
term in text
|
| 792 |
+
for term in ("one-time", "/month", "in total", "lifetime access")
|
| 793 |
+
)
|
| 794 |
+
|
| 795 |
+
if field == "lesson_count":
|
| 796 |
+
if re.search(r"\bpreview\b|\bfree\s+lessons?\b", lowered_query):
|
| 797 |
+
preview_count_patterns = {
|
| 798 |
+
"full-stack-ai-engineering": re.compile(
|
| 799 |
+
r"\b(?:first|try|explore)\s+6\s+lessons\b|\bfirst\s+six\s+lessons\b",
|
| 800 |
+
re.IGNORECASE,
|
| 801 |
+
),
|
| 802 |
+
"agent-engineering": re.compile(
|
| 803 |
+
r"\b(?:first|try|explore)\s+7\s+lessons\b|\bfirst\s+seven\s+lessons\b",
|
| 804 |
+
re.IGNORECASE,
|
| 805 |
+
),
|
| 806 |
+
}
|
| 807 |
+
parent_pattern = preview_count_patterns.get(offer_id)
|
| 808 |
+
if parent_pattern is not None:
|
| 809 |
+
# The parent-page exception is count-only. Some CTA DOM blocks
|
| 810 |
+
# combine "Try 7 Lessons Free" with paid lifetime access and a
|
| 811 |
+
# guarantee; never admit that mixed block as preview evidence.
|
| 812 |
+
mixed_paid_entitlement = bool(
|
| 813 |
+
re.search(
|
| 814 |
+
r"\b(?:lifetime|guarantee|refund|certificat(?:e|ion)|"
|
| 815 |
+
r"money[- ]back)\b|[$€£¥]\s*\d",
|
| 816 |
+
text,
|
| 817 |
+
)
|
| 818 |
+
)
|
| 819 |
+
return bool(parent_pattern.search(span_text)) and not (
|
| 820 |
+
mixed_paid_entitlement
|
| 821 |
+
)
|
| 822 |
+
pattern = TOTAL_LESSON_PATTERNS.get(offer_id)
|
| 823 |
+
return bool(pattern and pattern.search(span_text))
|
| 824 |
+
|
| 825 |
+
if field == "product_count":
|
| 826 |
+
return bool(re.search(r"\b\d+\s+products\b", text))
|
| 827 |
+
|
| 828 |
+
if field == "page_count":
|
| 829 |
+
return bool(re.search(r"\b\d[\d,]*[- ]page\b", text))
|
| 830 |
+
|
| 831 |
+
if field == "duration":
|
| 832 |
+
pattern = DURATION_PATTERNS.get(offer_id)
|
| 833 |
+
return bool(pattern and pattern.search(span_text))
|
| 834 |
+
|
| 835 |
+
if field == "prerequisite":
|
| 836 |
+
return any(
|
| 837 |
+
term in text
|
| 838 |
+
for term in (
|
| 839 |
+
"prerequisite",
|
| 840 |
+
"prior experience",
|
| 841 |
+
"no prior",
|
| 842 |
+
"no code",
|
| 843 |
+
"no coding",
|
| 844 |
+
"basic python",
|
| 845 |
+
"intermediate python",
|
| 846 |
+
"complete beginner",
|
| 847 |
+
"zero prior",
|
| 848 |
+
"experience with",
|
| 849 |
+
)
|
| 850 |
+
)
|
| 851 |
+
|
| 852 |
+
if field == "certificate":
|
| 853 |
+
if not re.search(r"\bcertificat(?:e|ion|ed)\b", text):
|
| 854 |
+
return False
|
| 855 |
+
return not is_preview or any(term in text for term in ("free", "preview"))
|
| 856 |
+
|
| 857 |
+
if field == "refund":
|
| 858 |
+
if offer_id == "building-llms-for-production":
|
| 859 |
+
return False
|
| 860 |
+
if offer_id == "mentorship":
|
| 861 |
+
if monthly_plan_intent(lowered_query):
|
| 862 |
+
return (
|
| 863 |
+
"monthly mentorship can be cancelled" in text
|
| 864 |
+
or "yearly plan carries a 30-day money-back guarantee" in text
|
| 865 |
+
or "money-back on yearly" in text
|
| 866 |
+
)
|
| 867 |
+
if yearly_plan_intent(lowered_query):
|
| 868 |
+
return "yearly" in text and "money-back" in text
|
| 869 |
+
return "yearly" in text and "money-back" in text
|
| 870 |
+
return "refund" in text
|
| 871 |
+
|
| 872 |
+
if field == "guarantee":
|
| 873 |
+
if re.search(r"\b(?:job|role|internship|placement|career)\b", lowered_query):
|
| 874 |
+
return bool(
|
| 875 |
+
re.search(r"\b(?:job|role|internship|placement|career|pathway)\b", text)
|
| 876 |
+
and re.search(r"\b(?:guarantee|promised|earned)\b", text)
|
| 877 |
+
)
|
| 878 |
+
return _span_supports_fact(
|
| 879 |
+
span_text, field="refund", offer_id=offer_id, query=query
|
| 880 |
+
)
|
| 881 |
+
|
| 882 |
+
if field == "discount":
|
| 883 |
+
if offer_id == "mentorship":
|
| 884 |
+
alpha_intent = bool(re.search(r"\b(?:new courses?|alpha)\b", lowered_query))
|
| 885 |
+
plan_intent = bool(
|
| 886 |
+
monthly_plan_intent(lowered_query)
|
| 887 |
+
or yearly_plan_intent(lowered_query)
|
| 888 |
+
or re.search(r"\b(?:billing|plan|save|savings?)\b|\b24%", lowered_query)
|
| 889 |
+
)
|
| 890 |
+
course_intent = bool(
|
| 891 |
+
re.search(
|
| 892 |
+
r"\b(?:courses?|full stack|agent(?:ic)? engineering|"
|
| 893 |
+
r"master ai for work|ai for work|existing)\b",
|
| 894 |
+
lowered_query,
|
| 895 |
+
)
|
| 896 |
+
)
|
| 897 |
+
if sum((alpha_intent, plan_intent, course_intent)) > 1:
|
| 898 |
+
return False
|
| 899 |
+
if alpha_intent:
|
| 900 |
+
return "alpha access" in text and "% off" in text
|
| 901 |
+
if plan_intent:
|
| 902 |
+
if monthly_plan_intent(lowered_query) and not yearly_plan_intent(
|
| 903 |
+
lowered_query
|
| 904 |
+
):
|
| 905 |
+
return False
|
| 906 |
+
return "save 24%" in text or (
|
| 907 |
+
"24% less" in text and "month to month" in text
|
| 908 |
+
)
|
| 909 |
+
if course_intent:
|
| 910 |
+
return "25% off" in text
|
| 911 |
+
return False
|
| 912 |
+
if offer_id in BUNDLE_OFFER_IDS:
|
| 913 |
+
if re.search(r"\bstudent\b|\bprevious\b|\bbuyer\b", lowered_query):
|
| 914 |
+
return False
|
| 915 |
+
if re.search(r"\bgroup\b|\bteam\b|\btwo or more\b", lowered_query):
|
| 916 |
+
return "group" in text and "bundle pricing" in text
|
| 917 |
+
return "bundle price" in text and "was" in text
|
| 918 |
+
return "discount" in text or (
|
| 919 |
+
"students get 50% off" in text
|
| 920 |
+
and any(term in text for term in ("previous", "groups of two"))
|
| 921 |
+
)
|
| 922 |
+
|
| 923 |
+
if field == "access":
|
| 924 |
+
permanent_access_requested = bool(PERMANENT_ACCESS_RE.search(lowered_query))
|
| 925 |
+
if offer_id == "mentorship":
|
| 926 |
+
if re.search(r"\bcancel|\bcancell|\bafter\b", lowered_query):
|
| 927 |
+
return (
|
| 928 |
+
"course access is active while" in text
|
| 929 |
+
or "if you cancel" in text
|
| 930 |
+
)
|
| 931 |
+
if permanent_access_requested:
|
| 932 |
+
return bool(PERMANENT_ACCESS_RE.search(text))
|
| 933 |
+
return any(
|
| 934 |
+
term in text
|
| 935 |
+
for term in ("included from day one", "course access", "alpha access")
|
| 936 |
+
)
|
| 937 |
+
if is_preview:
|
| 938 |
+
relation = any(
|
| 939 |
+
term in text for term in ("access", "lifetime", "keep", "forever")
|
| 940 |
+
)
|
| 941 |
+
preview_qualified = any(
|
| 942 |
+
term in text for term in ("free", "preview", "lessons", "no card")
|
| 943 |
+
)
|
| 944 |
+
permanent_qualified = not permanent_access_requested or bool(
|
| 945 |
+
PERMANENT_ACCESS_RE.search(text)
|
| 946 |
+
)
|
| 947 |
+
return relation and preview_qualified and permanent_qualified
|
| 948 |
+
if permanent_access_requested:
|
| 949 |
+
return bool(PERMANENT_ACCESS_RE.search(text))
|
| 950 |
+
return any(term in text for term in ("access", "lifetime", "included"))
|
| 951 |
+
|
| 952 |
+
if field == "inclusion":
|
| 953 |
+
if offer_id == "mentorship":
|
| 954 |
+
return any(
|
| 955 |
+
term in text
|
| 956 |
+
for term in ("included from day one", "25% off", "alpha access")
|
| 957 |
+
)
|
| 958 |
+
return "included" in text or "comes with" in text
|
| 959 |
+
|
| 960 |
+
return False
|
| 961 |
+
|
| 962 |
+
|
| 963 |
+
def evidence_span_supports_field(
|
| 964 |
+
span_text: str, *, field: str, offer_id: str, query: str
|
| 965 |
+
) -> bool:
|
| 966 |
+
"""Public validator counterpart to the typed retrieval evidence gate."""
|
| 967 |
+
|
| 968 |
+
return _span_supports_fact(
|
| 969 |
+
span_text,
|
| 970 |
+
field=field,
|
| 971 |
+
offer_id=offer_id,
|
| 972 |
+
query=query,
|
| 973 |
+
)
|
| 974 |
+
|
| 975 |
+
|
| 976 |
+
_CURRENCY_TOKEN_RE = re.compile(r"[$€£¥]\s*\d")
|
| 977 |
+
|
| 978 |
+
|
| 979 |
+
def _restrict_chunk_evidence(
|
| 980 |
+
chunk: dict[str, Any], query: str, fields: frozenset[str]
|
| 981 |
+
) -> dict[str, Any] | None:
|
| 982 |
+
offer_id = str(chunk.get("offer_id", ""))
|
| 983 |
+
allowed_fields = {
|
| 984 |
+
field
|
| 985 |
+
for field in fields
|
| 986 |
+
if offer_id in evidence_offer_ids_for_field(query, field)
|
| 987 |
+
}
|
| 988 |
+
if not allowed_fields:
|
| 989 |
+
return None
|
| 990 |
+
spans = [
|
| 991 |
+
span
|
| 992 |
+
for span in chunk.get("evidence_spans", [])
|
| 993 |
+
if isinstance(span, dict)
|
| 994 |
+
and any(
|
| 995 |
+
_span_supports_fact(
|
| 996 |
+
str(span.get("text", "")),
|
| 997 |
+
field=field,
|
| 998 |
+
offer_id=offer_id,
|
| 999 |
+
query=query,
|
| 1000 |
+
)
|
| 1001 |
+
for field in allowed_fields
|
| 1002 |
+
)
|
| 1003 |
+
]
|
| 1004 |
+
if not spans:
|
| 1005 |
+
return None
|
| 1006 |
+
restricted = dict(chunk)
|
| 1007 |
+
restricted["evidence_spans"] = spans
|
| 1008 |
+
return restricted
|
| 1009 |
|
| 1010 |
|
| 1011 |
def normalized_path(url: str) -> tuple[str, str]:
|
|
|
|
| 1017 |
|
| 1018 |
def allowed_paths_by_host() -> dict[str, list[str]]:
|
| 1019 |
result: dict[str, set[str]] = {}
|
| 1020 |
+
for page in all_pages():
|
| 1021 |
+
host = str(
|
| 1022 |
+
page.get("host") or urlparse(str(page.get("url", ""))).hostname or ""
|
| 1023 |
+
).lower()
|
| 1024 |
+
path = str(page.get("path") or urlparse(str(page.get("url", ""))).path or "/")
|
| 1025 |
+
path = path.rstrip("/") or "/"
|
| 1026 |
if host:
|
| 1027 |
result.setdefault(host, set()).add(path)
|
| 1028 |
+
if host in {"towardsai.com", "towardsai.net"}:
|
| 1029 |
+
result.setdefault(f"www.{host}", set()).add(path)
|
| 1030 |
+
for host, paths in LEGACY_PUBLIC_PATHS_BY_HOST.items():
|
| 1031 |
+
result.setdefault(host, set()).update(paths)
|
| 1032 |
+
result.setdefault(f"www.{host}", set()).update(paths)
|
| 1033 |
return {host: sorted(paths) for host, paths in result.items()}
|
| 1034 |
|
| 1035 |
|
| 1036 |
def page_is_allowed(url: str) -> bool:
|
| 1037 |
host, path = normalized_path(url)
|
| 1038 |
+
parsed = urlparse(url)
|
| 1039 |
if not host:
|
| 1040 |
return False
|
| 1041 |
if path.startswith(("/courses/take", "/enroll", "/order", "/checkout", "/cart")):
|
| 1042 |
return False
|
| 1043 |
+
if path.startswith(
|
| 1044 |
+
(
|
| 1045 |
+
"/users",
|
| 1046 |
+
"/account",
|
| 1047 |
+
"/admin",
|
| 1048 |
+
"/wp-admin",
|
| 1049 |
+
"/wp-login",
|
| 1050 |
+
"/wp-json",
|
| 1051 |
+
"/wp-content",
|
| 1052 |
+
"/wp-includes",
|
| 1053 |
+
"/xmlrpc.php",
|
| 1054 |
+
)
|
| 1055 |
+
):
|
| 1056 |
return False
|
| 1057 |
+
if "preview" in parse_qs(parsed.query):
|
| 1058 |
+
return False
|
| 1059 |
+
if host in {item.lower() for item in settings.site_wide_hosts}:
|
| 1060 |
+
return True
|
| 1061 |
allowed = allowed_paths_by_host()
|
| 1062 |
return path in allowed.get(host, [])
|
| 1063 |
|
| 1064 |
|
| 1065 |
def source_for_url(url: str) -> dict[str, Any] | None:
|
| 1066 |
+
key = _canonical_key(url)
|
| 1067 |
for page in pages():
|
| 1068 |
+
if _canonical_key(str(page.get("url", ""))) == key:
|
|
|
|
|
|
|
| 1069 |
return page
|
| 1070 |
return None
|
| 1071 |
|
| 1072 |
|
| 1073 |
+
def chunks() -> tuple[dict[str, Any], ...]:
|
| 1074 |
+
result: list[dict[str, Any]] = []
|
| 1075 |
+
for page in pages():
|
| 1076 |
+
raw_chunks = page["chunks"]
|
| 1077 |
+
for raw_chunk in raw_chunks:
|
| 1078 |
+
text = SPACE_RE.sub(" ", str(raw_chunk.get("text", ""))).strip()
|
| 1079 |
+
url = str(page.get("url", ""))
|
| 1080 |
+
heading = str(raw_chunk.get("heading", "")).strip()
|
| 1081 |
+
chunk_id = str(raw_chunk.get("chunk_id", "")).strip()
|
| 1082 |
+
result.append(
|
| 1083 |
+
{
|
| 1084 |
+
"chunk_id": chunk_id,
|
| 1085 |
+
"title": str(page.get("title", "")),
|
| 1086 |
+
"url": url,
|
| 1087 |
+
"host": str(page.get("host", "")),
|
| 1088 |
+
"path": str(page.get("path", "")),
|
| 1089 |
+
"kind": str(page.get("kind", "page")),
|
| 1090 |
+
"offer_id": str(page.get("offer_id", "")),
|
| 1091 |
+
"entity_id": str(page.get("entity_id", "")),
|
| 1092 |
+
"heading": heading,
|
| 1093 |
+
"headings": [heading] if heading else [],
|
| 1094 |
+
"text": text,
|
| 1095 |
+
"chunk_index": raw_chunk.get("index"),
|
| 1096 |
+
"evidence_spans": list(raw_chunk.get("evidence_spans", [])),
|
| 1097 |
+
"authority": page.get("authority", _authority(page)),
|
| 1098 |
+
"fetched_at": page.get("fetched_at")
|
| 1099 |
+
or page.get("catalog_generated_at", ""),
|
| 1100 |
+
"status": page.get("status", "active"),
|
| 1101 |
+
}
|
| 1102 |
+
)
|
| 1103 |
+
return tuple(result)
|
| 1104 |
+
|
| 1105 |
+
|
| 1106 |
+
def _expanded_query_tokens(query: str) -> list[str]:
|
| 1107 |
+
tokens = _token_list(query)
|
| 1108 |
+
expansions = {
|
| 1109 |
+
"mentor": ("mentorship",),
|
| 1110 |
+
"mentorship": ("mentor",),
|
| 1111 |
+
"cost": ("price", "pricing"),
|
| 1112 |
+
"price": ("cost", "pricing"),
|
| 1113 |
+
"classes": ("course", "courses"),
|
| 1114 |
+
"class": ("course",),
|
| 1115 |
+
"company": ("enterprise", "team"),
|
| 1116 |
+
"business": ("enterprise", "company"),
|
| 1117 |
+
"certificate": ("certification",),
|
| 1118 |
+
}
|
| 1119 |
+
for token in tuple(tokens):
|
| 1120 |
+
tokens.extend(expansions.get(token, ()))
|
| 1121 |
+
return tokens
|
| 1122 |
+
|
| 1123 |
+
|
| 1124 |
+
def _routing_boost(query: str, chunk: dict[str, Any]) -> float:
|
| 1125 |
+
lowered = query.lower()
|
| 1126 |
+
url = str(chunk.get("url", "")).lower()
|
| 1127 |
+
chunk_path = normalized_path(url)[1]
|
| 1128 |
+
chunk_index = chunk.get("chunk_index")
|
| 1129 |
+
kind = str(chunk.get("kind", ""))
|
| 1130 |
+
evidence_text = str(chunk.get("text", ""))
|
| 1131 |
+
evidence_lower = evidence_text.lower()
|
| 1132 |
+
score = 0.0
|
| 1133 |
+
|
| 1134 |
+
course_starter = "help deciding which course to take" in lowered
|
| 1135 |
+
canonical_learning_paths = {
|
| 1136 |
+
"/academy/full-stack-ai-engineering",
|
| 1137 |
+
"/academy/agent-engineering",
|
| 1138 |
+
"/academy/llm-primer",
|
| 1139 |
+
"/academy/python-for-ai-engineering",
|
| 1140 |
+
"/academy/ai-for-work",
|
| 1141 |
+
"/academy/building-llms-for-production",
|
| 1142 |
+
}
|
| 1143 |
+
if course_starter and chunk_path in canonical_learning_paths:
|
| 1144 |
+
score += 36.0 if chunk_index == 0 else 6.0
|
| 1145 |
+
|
| 1146 |
+
free_starter = "free resources to learn before committing" in lowered
|
| 1147 |
+
if free_starter:
|
| 1148 |
+
if kind in {"free_resource", "digital_download"}:
|
| 1149 |
+
score += 42.0 + (8.0 if chunk_index == 0 else 0.0)
|
| 1150 |
+
elif chunk_path == "/academy/book":
|
| 1151 |
+
score += 26.0
|
| 1152 |
+
|
| 1153 |
+
if "integrate ai into my company" in lowered:
|
| 1154 |
+
if chunk_path == "/valuecreation":
|
| 1155 |
+
score += 52.0 + (12.0 if chunk_index == 0 else 0.0)
|
| 1156 |
+
elif chunk_path == "/enterpriseenablement":
|
| 1157 |
+
score += 44.0 + (12.0 if chunk_index == 0 else 0.0)
|
| 1158 |
+
elif chunk_path.startswith("/enterprise/"):
|
| 1159 |
+
score += 28.0 + (8.0 if chunk_index == 0 else 0.0)
|
| 1160 |
+
|
| 1161 |
+
if "training inside my company" in lowered:
|
| 1162 |
+
if chunk_path == "/enterpriseenablement":
|
| 1163 |
+
score += 44.0
|
| 1164 |
+
elif chunk_path.startswith("/enterprise/"):
|
| 1165 |
+
score += 28.0
|
| 1166 |
+
|
| 1167 |
+
if chunk_path == "/academy/bundles/get-it-all" and any(
|
| 1168 |
+
phrase in lowered for phrase in ("best value", "get it all", "every course")
|
| 1169 |
+
):
|
| 1170 |
+
score += 12.0
|
| 1171 |
+
if chunk_path == "/academy/python-for-ai-engineering" and any(
|
| 1172 |
+
phrase in lowered
|
| 1173 |
+
for phrase in (
|
| 1174 |
+
"beginner",
|
| 1175 |
+
"non-coder",
|
| 1176 |
+
"non coder",
|
| 1177 |
+
"don't code",
|
| 1178 |
+
"do not code",
|
| 1179 |
+
)
|
| 1180 |
+
):
|
| 1181 |
+
score += 10.0
|
| 1182 |
+
if (
|
| 1183 |
+
chunk_path == "/enterpriseenablement"
|
| 1184 |
+
and "training" in lowered
|
| 1185 |
+
and any(phrase in lowered for phrase in ("company", "business", "team"))
|
| 1186 |
+
):
|
| 1187 |
+
score += 10.0
|
| 1188 |
+
if "help deciding which course" in lowered and chunk_path == "/academy":
|
| 1189 |
+
score += 5.0
|
| 1190 |
+
if (
|
| 1191 |
+
chunk_path == "/academy/mentorship"
|
| 1192 |
+
and any(term in lowered for term in ("mentor", "mentorship"))
|
| 1193 |
+
and any(term in lowered for term in ("course", "included", "include", "access"))
|
| 1194 |
+
and "llm fundamentals" in evidence_lower
|
| 1195 |
+
and "included from day one" in evidence_lower
|
| 1196 |
+
):
|
| 1197 |
+
score += 30.0
|
| 1198 |
+
|
| 1199 |
+
wants_free_preview = "free" in lowered and any(
|
| 1200 |
+
term in lowered for term in ("preview", "lesson")
|
| 1201 |
+
)
|
| 1202 |
+
named_offer_routes = (
|
| 1203 |
+
("full stack", "/academy/full-stack-ai-engineering", 16.0),
|
| 1204 |
+
("agent engineering", "/academy/agent-engineering", 16.0),
|
| 1205 |
+
("llm fundamentals", "/academy/llm-primer", 16.0),
|
| 1206 |
+
("llm primer", "/academy/llm-primer", 16.0),
|
| 1207 |
+
("python", "/academy/python-for-ai-engineering", 16.0),
|
| 1208 |
+
("master ai for work", "/academy/ai-for-work", 16.0),
|
| 1209 |
+
("ai for work", "/academy/ai-for-work", 16.0),
|
| 1210 |
+
(
|
| 1211 |
+
"from non-coder to ai engineer",
|
| 1212 |
+
"/academy/bundles/from-coding-novice-to-advanced-llm-developer",
|
| 1213 |
+
18.0,
|
| 1214 |
+
),
|
| 1215 |
+
(
|
| 1216 |
+
"from developer to advanced ai engineer",
|
| 1217 |
+
"/academy/bundles/10-hour-crash-course-into-llm-developer-expert",
|
| 1218 |
+
18.0,
|
| 1219 |
+
),
|
| 1220 |
+
)
|
| 1221 |
+
for phrase, target_path, boost in named_offer_routes:
|
| 1222 |
+
if target_path in {
|
| 1223 |
+
"/academy/full-stack-ai-engineering",
|
| 1224 |
+
"/academy/agent-engineering",
|
| 1225 |
+
} and (wants_free_preview or "webinar" in lowered):
|
| 1226 |
+
continue
|
| 1227 |
+
if phrase in lowered and chunk_path == target_path:
|
| 1228 |
+
score += boost
|
| 1229 |
+
if (
|
| 1230 |
+
"full stack" in lowered
|
| 1231 |
+
and wants_free_preview
|
| 1232 |
+
and chunk_path == "/academy/full-stack-ai-engineering-free-preview"
|
| 1233 |
+
):
|
| 1234 |
+
score += 24.0
|
| 1235 |
+
if (
|
| 1236 |
+
"agent engineering" in lowered
|
| 1237 |
+
and wants_free_preview
|
| 1238 |
+
and chunk_path == "/academy/agent-engineering-free-preview"
|
| 1239 |
+
):
|
| 1240 |
+
score += 24.0
|
| 1241 |
+
if "building llms" in lowered:
|
| 1242 |
+
wants_resources = any(term in lowered for term in ("resource", "companion"))
|
| 1243 |
+
target_path = (
|
| 1244 |
+
"/academy/book"
|
| 1245 |
+
if wants_resources
|
| 1246 |
+
else "/academy/building-llms-for-production"
|
| 1247 |
+
)
|
| 1248 |
+
if chunk_path == target_path:
|
| 1249 |
+
score += 18.0
|
| 1250 |
+
if "webinar" in lowered and chunk_path == "/webinars/agentengineering":
|
| 1251 |
+
score += 40.0
|
| 1252 |
+
if (
|
| 1253 |
+
any(term in lowered for term in ("price", "cost", "how much"))
|
| 1254 |
+
and "$" in evidence_text
|
| 1255 |
+
):
|
| 1256 |
+
score += 10.0
|
| 1257 |
+
if (
|
| 1258 |
+
"lesson" in lowered
|
| 1259 |
+
and any(phrase in lowered for phrase in ("how many", "number of"))
|
| 1260 |
+
and re.search(r"\b\d[\d,+]*\s+lessons\b", evidence_lower)
|
| 1261 |
+
):
|
| 1262 |
+
score += 16.0
|
| 1263 |
+
if (
|
| 1264 |
+
"page" in lowered
|
| 1265 |
+
and any(phrase in lowered for phrase in ("how many", "number of"))
|
| 1266 |
+
and re.search(r"\b\d[\d,]*[- ]page\b", evidence_lower)
|
| 1267 |
+
):
|
| 1268 |
+
score += 16.0
|
| 1269 |
+
if (
|
| 1270 |
+
"product" in lowered
|
| 1271 |
+
and any(phrase in lowered for phrase in ("how many", "number of"))
|
| 1272 |
+
and re.search(r"\b\d+\s+products\b", evidence_lower)
|
| 1273 |
+
):
|
| 1274 |
+
score += 16.0
|
| 1275 |
+
|
| 1276 |
+
rules = (
|
| 1277 |
+
(("mentor", "mentorship"), "/academy/mentorship/", 7.0),
|
| 1278 |
+
(
|
| 1279 |
+
("bundle", "best value", "every course", "get it all"),
|
| 1280 |
+
"/academy/bundles/",
|
| 1281 |
+
4.0,
|
| 1282 |
+
),
|
| 1283 |
+
(
|
| 1284 |
+
("codex", "claude", "coding agent"),
|
| 1285 |
+
"/enterprise/agentic-developer-conversion/",
|
| 1286 |
+
24.0,
|
| 1287 |
+
),
|
| 1288 |
+
(
|
| 1289 |
+
("software developer", "developers into ai engineers"),
|
| 1290 |
+
"/enterprise/software-developer-to-ai-engineer/",
|
| 1291 |
+
24.0,
|
| 1292 |
+
),
|
| 1293 |
+
(
|
| 1294 |
+
("consulting", "deployment", "value creation", "private equity"),
|
| 1295 |
+
"/valuecreation/",
|
| 1296 |
+
20.0,
|
| 1297 |
+
),
|
| 1298 |
+
(("enablement", "enterprise academy"), "/enterpriseenablement/", 20.0),
|
| 1299 |
+
)
|
| 1300 |
+
for phrases, path, boost in rules:
|
| 1301 |
+
target_path = path.rstrip("/") or "/"
|
| 1302 |
+
path_matches = (
|
| 1303 |
+
chunk_path.startswith(target_path + "/")
|
| 1304 |
+
if target_path == "/academy/bundles"
|
| 1305 |
+
else chunk_path == target_path
|
| 1306 |
+
)
|
| 1307 |
+
if any(phrase in lowered for phrase in phrases) and path_matches:
|
| 1308 |
+
score += boost
|
| 1309 |
+
return score
|
| 1310 |
+
|
| 1311 |
+
|
| 1312 |
+
def retrieve(
|
| 1313 |
+
query: str, *, current_url: str = "", limit: int = 7
|
| 1314 |
+
) -> list[dict[str, Any]]:
|
| 1315 |
+
"""Retrieve fresh evidence chunks. An empty result means the helper must abstain."""
|
| 1316 |
+
query_terms = _expanded_query_tokens(query)
|
| 1317 |
+
if not query_terms or limit <= 0:
|
| 1318 |
+
return []
|
| 1319 |
+
corpus = list(chunks())
|
| 1320 |
+
if not corpus:
|
| 1321 |
+
return []
|
| 1322 |
+
|
| 1323 |
+
normalized_query = SPACE_RE.sub(" ", query.strip().lower()).rstrip(".")
|
| 1324 |
+
course_paths = {
|
| 1325 |
+
"/academy/full-stack-ai-engineering",
|
| 1326 |
+
"/academy/agent-engineering",
|
| 1327 |
+
"/academy/llm-primer",
|
| 1328 |
+
"/academy/python-for-ai-engineering",
|
| 1329 |
+
"/academy/ai-for-work",
|
| 1330 |
+
"/academy/building-llms-for-production",
|
| 1331 |
+
}
|
| 1332 |
+
if normalized_query == "i want help deciding which course to take":
|
| 1333 |
+
corpus = [chunk for chunk in corpus if chunk.get("path") in course_paths]
|
| 1334 |
+
elif normalized_query == "i want help to integrate ai into my company":
|
| 1335 |
+
corpus = [
|
| 1336 |
+
chunk
|
| 1337 |
+
for chunk in corpus
|
| 1338 |
+
if chunk.get("path") in {"/valuecreation", "/enterpriseenablement"}
|
| 1339 |
+
or str(chunk.get("path", "")).startswith("/enterprise/")
|
| 1340 |
+
]
|
| 1341 |
+
elif normalized_query == "i want a training inside my company":
|
| 1342 |
+
corpus = [
|
| 1343 |
+
chunk
|
| 1344 |
+
for chunk in corpus
|
| 1345 |
+
if chunk.get("path") == "/enterpriseenablement"
|
| 1346 |
+
or str(chunk.get("path", "")).startswith("/enterprise/")
|
| 1347 |
+
]
|
| 1348 |
+
elif normalized_query == (
|
| 1349 |
+
"i'm looking for more free resources to learn before committing to buying "
|
| 1350 |
+
"a course"
|
| 1351 |
+
):
|
| 1352 |
+
corpus = [
|
| 1353 |
+
chunk
|
| 1354 |
+
for chunk in corpus
|
| 1355 |
+
if chunk.get("kind") in {"free_resource", "digital_download"}
|
| 1356 |
+
or chunk.get("path") == "/academy/book"
|
| 1357 |
+
]
|
| 1358 |
+
elif normalized_query == "i want to find mentors":
|
| 1359 |
+
corpus = [
|
| 1360 |
+
chunk for chunk in corpus if chunk.get("path") == "/academy/mentorship"
|
| 1361 |
+
]
|
| 1362 |
+
|
| 1363 |
+
target_offer_ids = offer_ids_for_query(query)
|
| 1364 |
+
evidence_offer_ids = evidence_offer_ids_for_query(query)
|
| 1365 |
+
if evidence_offer_ids:
|
| 1366 |
+
corpus = [
|
| 1367 |
+
chunk
|
| 1368 |
+
for chunk in corpus
|
| 1369 |
+
if str(chunk.get("offer_id", "")) in evidence_offer_ids
|
| 1370 |
+
]
|
| 1371 |
+
fact_fields = requested_fact_fields(query)
|
| 1372 |
+
if fact_fields:
|
| 1373 |
+
# High-risk facts without an explicit offer are ambiguous by definition.
|
| 1374 |
+
# Ask the visitor to contact the team instead of mixing offer policies.
|
| 1375 |
+
if not target_offer_ids:
|
| 1376 |
+
return []
|
| 1377 |
+
if mixed_preview_fact_request(query):
|
| 1378 |
+
return []
|
| 1379 |
+
restricted_corpus: list[dict[str, Any]] = []
|
| 1380 |
+
for chunk in corpus:
|
| 1381 |
+
restricted = _restrict_chunk_evidence(chunk, query, fact_fields)
|
| 1382 |
+
if restricted is not None:
|
| 1383 |
+
restricted_corpus.append(restricted)
|
| 1384 |
+
corpus = restricted_corpus
|
| 1385 |
+
if not corpus:
|
| 1386 |
+
return []
|
| 1387 |
|
| 1388 |
+
document_terms = [_token_list(str(chunk.get("text", ""))) for chunk in corpus]
|
| 1389 |
+
document_frequency: Counter[str] = Counter()
|
| 1390 |
+
for terms in document_terms:
|
| 1391 |
+
document_frequency.update(set(terms))
|
| 1392 |
+
average_length = sum(map(len, document_terms)) / max(len(document_terms), 1)
|
| 1393 |
+
query_counts = Counter(query_terms)
|
| 1394 |
+
query_term_set = set(query_counts)
|
| 1395 |
+
informative_query_terms = {
|
| 1396 |
+
term
|
| 1397 |
+
for term in query_term_set - GENERIC_RETRIEVAL_TERMS
|
| 1398 |
+
if not term.replace(".", "", 1).isdigit()
|
| 1399 |
}
|
| 1400 |
+
current_key = _canonical_key(current_url)
|
| 1401 |
+
scored: list[tuple[float, int, dict[str, Any]]] = []
|
| 1402 |
|
| 1403 |
+
for index, (chunk, terms) in enumerate(zip(corpus, document_terms, strict=True)):
|
| 1404 |
+
counts = Counter(terms)
|
| 1405 |
+
matched = set(query_counts) & set(counts)
|
| 1406 |
+
searchable = " ".join(
|
| 1407 |
[
|
| 1408 |
+
str(chunk.get("title", "")),
|
| 1409 |
+
str(chunk.get("heading", "")),
|
| 1410 |
+
str(chunk.get("path", "")),
|
|
|
|
|
|
|
| 1411 |
]
|
| 1412 |
)
|
| 1413 |
+
metadata_tokens = tokenize(searchable)
|
| 1414 |
+
metadata_matches = set(query_counts) & metadata_tokens
|
| 1415 |
+
all_matches = matched | metadata_matches
|
| 1416 |
+
route_boost = _routing_boost(query, chunk)
|
| 1417 |
+
if fact_fields and target_offer_ids:
|
| 1418 |
+
# Typed evidence has already passed the strict offer/field/qualifier
|
| 1419 |
+
# gate. It should not be discarded merely because the visitor used
|
| 1420 |
+
# a synonym such as "cancelling" while the page says "cancel".
|
| 1421 |
+
route_boost += 20.0
|
| 1422 |
+
if not all_matches and route_boost < 10.0:
|
| 1423 |
+
continue
|
| 1424 |
+
informative_matches = all_matches & informative_query_terms
|
| 1425 |
+
trusted_route = route_boost >= 10.0
|
| 1426 |
+
if informative_query_terms and not informative_matches and not trusted_route:
|
| 1427 |
+
continue
|
| 1428 |
+
informative_coverage = len(informative_matches) / max(
|
| 1429 |
+
len(informative_query_terms), 1
|
| 1430 |
+
)
|
| 1431 |
+
if (
|
| 1432 |
+
informative_query_terms
|
| 1433 |
+
and informative_coverage < 0.67
|
| 1434 |
+
and not trusted_route
|
| 1435 |
+
):
|
| 1436 |
+
continue
|
| 1437 |
+
|
| 1438 |
+
score = 0.0
|
| 1439 |
+
length = max(len(terms), 1)
|
| 1440 |
+
for term, query_frequency in query_counts.items():
|
| 1441 |
+
frequency = counts.get(term, 0)
|
| 1442 |
+
if not frequency:
|
| 1443 |
+
continue
|
| 1444 |
+
frequency_docs = document_frequency.get(term, 0)
|
| 1445 |
+
inverse_frequency = math.log(
|
| 1446 |
+
1 + (len(corpus) - frequency_docs + 0.5) / (frequency_docs + 0.5)
|
| 1447 |
+
)
|
| 1448 |
+
denominator = frequency + 1.2 * (
|
| 1449 |
+
0.25 + 0.75 * length / max(average_length, 1)
|
| 1450 |
+
)
|
| 1451 |
+
score += (
|
| 1452 |
+
inverse_frequency
|
| 1453 |
+
* (frequency * 2.2 / denominator)
|
| 1454 |
+
* min(query_frequency, 2)
|
| 1455 |
+
)
|
| 1456 |
+
score += 1.4 * len(metadata_matches)
|
| 1457 |
+
score += 2.0 * len(all_matches)
|
| 1458 |
+
# In a tiny corpus BM25's inverse-frequency value is small even for an
|
| 1459 |
+
# exact match. Reward multiple distinct, non-generic matches while a
|
| 1460 |
+
# lone generic word such as "course" still cannot retrieve a page.
|
| 1461 |
+
if len(informative_matches) >= 2:
|
| 1462 |
+
score += 0.75 * len(informative_matches)
|
| 1463 |
+
score += route_boost
|
| 1464 |
+
if current_key == _canonical_key(str(chunk.get("url", ""))):
|
| 1465 |
+
score += 1.0
|
| 1466 |
+
score *= 0.85 + min(_authority(chunk), 5.0) * 0.05
|
| 1467 |
+
if score >= 1.0:
|
| 1468 |
+
scored.append((score, index, chunk))
|
| 1469 |
+
|
| 1470 |
+
scored.sort(key=lambda item: (-item[0], item[1]))
|
| 1471 |
+
result: list[dict[str, Any]] = []
|
| 1472 |
+
per_url: defaultdict[str, int] = defaultdict(int)
|
| 1473 |
+
diversify = normalized_query in {
|
| 1474 |
+
"i want help deciding which course to take",
|
| 1475 |
+
"i want help to integrate ai into my company",
|
| 1476 |
+
"i'm looking for more free resources to learn before committing to buying a course",
|
| 1477 |
+
}
|
| 1478 |
+
per_url_limit = 1 if diversify else 2
|
| 1479 |
+
for _score, _index, chunk in scored:
|
| 1480 |
+
url = str(chunk.get("url", ""))
|
| 1481 |
+
if per_url[url] >= per_url_limit:
|
| 1482 |
+
continue
|
| 1483 |
+
result.append(chunk)
|
| 1484 |
+
per_url[url] += 1
|
| 1485 |
+
if len(result) >= limit:
|
| 1486 |
+
break
|
| 1487 |
+
return result
|
| 1488 |
+
|
| 1489 |
+
|
| 1490 |
+
def sources_from_pages(
|
| 1491 |
+
selected: list[dict[str, Any]], limit: int = 4
|
| 1492 |
+
) -> list[dict[str, str]]:
|
| 1493 |
result = []
|
| 1494 |
for page in selected:
|
| 1495 |
url = str(page.get("url", ""))
|
| 1496 |
+
if not url or url.startswith("internal://"):
|
| 1497 |
continue
|
| 1498 |
result.append(
|
| 1499 |
{
|
|
|
|
| 1512 |
|
| 1513 |
|
| 1514 |
def in_scope(text: str, history: list[str] | None = None) -> bool:
|
| 1515 |
+
lowered = text.lower()
|
| 1516 |
allowed_terms = {
|
| 1517 |
"course",
|
| 1518 |
"courses",
|
|
|
|
| 1524 |
"training",
|
| 1525 |
"company",
|
| 1526 |
"business",
|
| 1527 |
+
"beginner",
|
| 1528 |
+
"code",
|
| 1529 |
+
"coder",
|
| 1530 |
+
"coding",
|
| 1531 |
"team",
|
| 1532 |
"consulting",
|
| 1533 |
+
"developer",
|
| 1534 |
"resource",
|
| 1535 |
"resources",
|
| 1536 |
"youtube",
|
|
|
|
| 1544 |
"agents",
|
| 1545 |
"genai",
|
| 1546 |
"certificate",
|
| 1547 |
+
"certification",
|
| 1548 |
+
"preview",
|
| 1549 |
+
"lesson",
|
| 1550 |
+
"lessons",
|
| 1551 |
+
"free",
|
| 1552 |
+
"full stack",
|
| 1553 |
+
"ebook",
|
| 1554 |
+
"e-book",
|
| 1555 |
+
"lifetime",
|
| 1556 |
+
"access",
|
| 1557 |
+
"feature",
|
| 1558 |
+
"features",
|
| 1559 |
+
"duration",
|
| 1560 |
+
"hour",
|
| 1561 |
+
"hours",
|
| 1562 |
+
"guarantee",
|
| 1563 |
+
"support",
|
| 1564 |
"refund",
|
| 1565 |
"price",
|
| 1566 |
+
"pricing",
|
| 1567 |
+
"cost",
|
| 1568 |
"coupon",
|
| 1569 |
"discount",
|
| 1570 |
+
"engineer",
|
| 1571 |
+
"experience",
|
| 1572 |
+
"founder",
|
| 1573 |
"promo",
|
| 1574 |
+
"professional",
|
| 1575 |
"career",
|
| 1576 |
}
|
| 1577 |
+
if any(term in lowered for term in allowed_terms):
|
| 1578 |
+
return True
|
| 1579 |
+
|
| 1580 |
+
# Conversation history is useful only for genuinely referential follow-ups.
|
| 1581 |
+
# It must never make a new, unrelated question appear in scope.
|
| 1582 |
+
followup_terms = {
|
| 1583 |
+
"it",
|
| 1584 |
+
"that",
|
| 1585 |
+
"this",
|
| 1586 |
+
"they",
|
| 1587 |
+
"them",
|
| 1588 |
+
"those",
|
| 1589 |
+
"one",
|
| 1590 |
+
"ones",
|
| 1591 |
+
"included",
|
| 1592 |
+
"access",
|
| 1593 |
+
"prerequisites",
|
| 1594 |
+
}
|
| 1595 |
+
raw_words = {token.lower() for token in WORD_RE.findall(lowered)}
|
| 1596 |
+
is_followup = len(raw_words) <= 10 and bool(raw_words & followup_terms)
|
| 1597 |
+
if not is_followup:
|
| 1598 |
+
return False
|
| 1599 |
+
prior = " ".join(history or []).lower()
|
| 1600 |
+
return any(term in prior for term in allowed_terms)
|
| 1601 |
|
| 1602 |
|
| 1603 |
def coupon_intent(text: str) -> bool:
|
|
|
|
| 1611 |
lowered.count(term) for term in ("coupon", "promo code", "discount code")
|
| 1612 |
)
|
| 1613 |
return prior_coupon_mentions >= 2 or any(
|
| 1614 |
+
term in text.lower()
|
| 1615 |
+
for term in ("please", "really", "need", "student", "can't")
|
| 1616 |
)
|
| 1617 |
+
|
| 1618 |
+
|
| 1619 |
+
def clear_catalog_caches() -> None:
|
| 1620 |
+
"""Clear catalog caches after an on-disk refresh or in tests."""
|
| 1621 |
+
_pages_payload_cached.cache_clear()
|
| 1622 |
+
assistant_notes.cache_clear()
|
tai_helper/llm.py
CHANGED
|
@@ -1,61 +1,530 @@
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
import time
|
| 4 |
-
|
| 5 |
-
from
|
|
|
|
| 6 |
|
|
|
|
| 7 |
from google import genai
|
| 8 |
|
| 9 |
-
from .catalog import
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
from .settings import settings
|
| 11 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
|
| 13 |
SYSTEM_INSTRUCTION = """You are Towards AI Helper, a concise public assistant for anonymous prospective students.
|
| 14 |
|
| 15 |
Your only job is to help users choose Towards AI courses, bundles, mentorship,
|
| 16 |
free resources, the book, community, or B2B training/consulting.
|
| 17 |
|
| 18 |
-
|
| 19 |
-
-
|
| 20 |
-
|
| 21 |
-
-
|
| 22 |
-
-
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
-
|
| 26 |
-
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
"""
|
| 28 |
|
| 29 |
|
| 30 |
@dataclass(frozen=True)
|
| 31 |
class LLMResult:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
answer: str
|
| 33 |
usage: dict[str, Any] = field(default_factory=dict)
|
| 34 |
latency_ms: int = 0
|
| 35 |
|
| 36 |
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
|
| 40 |
-
|
| 41 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 42 |
if str(page.get("url", "")).startswith("internal://"):
|
| 43 |
continue
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
|
| 52 |
-
|
| 53 |
-
|
| 54 |
-
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
|
| 58 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 59 |
|
| 60 |
|
| 61 |
def build_prompt(
|
|
@@ -72,36 +541,128 @@ def build_prompt(
|
|
| 72 |
)
|
| 73 |
return f"""The visitor is on a public Towards AI page.
|
| 74 |
|
|
|
|
| 75 |
Current URL: {current_url or "unknown"}
|
| 76 |
Current page title: {page_title or "unknown"}
|
| 77 |
Initial forced prompt: {selected_prompt or "unknown"}
|
| 78 |
|
| 79 |
-
Conversation so far:
|
| 80 |
{turns or "(none)"}
|
| 81 |
|
| 82 |
-
Visitor message:
|
| 83 |
{query}
|
|
|
|
| 84 |
|
| 85 |
-
Use only these public sales/resource sources and the routing notes:
|
| 86 |
{_context_block(selected_pages)}
|
| 87 |
|
| 88 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 89 |
"""
|
| 90 |
|
| 91 |
|
| 92 |
-
def
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 93 |
if not settings.gemini_api_key:
|
| 94 |
raise RuntimeError("GEMINI_API_KEY is not configured")
|
| 95 |
|
| 96 |
-
started = time.monotonic()
|
| 97 |
client = genai.Client(api_key=settings.gemini_api_key)
|
| 98 |
response = client.models.generate_content(
|
| 99 |
-
model=settings.
|
| 100 |
contents=prompt,
|
| 101 |
config={
|
| 102 |
"system_instruction": SYSTEM_INSTRUCTION,
|
| 103 |
-
"temperature":
|
| 104 |
"max_output_tokens": settings.max_output_tokens,
|
|
|
|
|
|
|
| 105 |
},
|
| 106 |
)
|
| 107 |
usage_metadata = getattr(response, "usage_metadata", None)
|
|
@@ -112,8 +673,573 @@ def generate_answer(prompt: str) -> LLMResult:
|
|
| 112 |
"output_tokens": getattr(usage_metadata, "candidates_token_count", None),
|
| 113 |
"total_tokens": getattr(usage_metadata, "total_token_count", None),
|
| 114 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 115 |
return LLMResult(
|
| 116 |
answer=(getattr(response, "text", "") or "").strip(),
|
| 117 |
-
usage=
|
| 118 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 119 |
)
|
|
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
| 3 |
+
import hashlib
|
| 4 |
+
import json
|
| 5 |
+
import logging
|
| 6 |
+
import re
|
| 7 |
import time
|
| 8 |
+
import unicodedata
|
| 9 |
+
from dataclasses import dataclass, field, replace
|
| 10 |
+
from typing import Any, Literal
|
| 11 |
|
| 12 |
+
import requests
|
| 13 |
from google import genai
|
| 14 |
|
| 15 |
+
from .catalog import (
|
| 16 |
+
evidence_offer_ids_for_field,
|
| 17 |
+
evidence_span_supports_field,
|
| 18 |
+
mixed_preview_fact_request,
|
| 19 |
+
monthly_plan_intent,
|
| 20 |
+
offer_alias_tokens,
|
| 21 |
+
offer_ids_for_query,
|
| 22 |
+
requested_fact_fields,
|
| 23 |
+
)
|
| 24 |
from .settings import settings
|
| 25 |
|
| 26 |
+
DEEPSEEK_PROVIDER = "deepseek"
|
| 27 |
+
GEMINI_PROVIDER = "google_genai"
|
| 28 |
+
DEFAULT_TEMPERATURE = 0.0
|
| 29 |
+
MAX_CONTEXT_CHARS = 18000
|
| 30 |
+
MAX_CHUNK_TEXT_CHARS = 4500
|
| 31 |
+
MAX_EVIDENCE_QUOTE_CHARS = 320
|
| 32 |
+
logger = logging.getLogger(__name__)
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
GROUNDING_RESPONSE_SCHEMA: dict[str, Any] = {
|
| 36 |
+
"type": "object",
|
| 37 |
+
"additionalProperties": False,
|
| 38 |
+
"required": ["status", "claims"],
|
| 39 |
+
"properties": {
|
| 40 |
+
"status": {"type": "string", "enum": ["answered", "not_found"]},
|
| 41 |
+
"claims": {
|
| 42 |
+
"type": "array",
|
| 43 |
+
"maxItems": 6,
|
| 44 |
+
"items": {
|
| 45 |
+
"type": "object",
|
| 46 |
+
"additionalProperties": False,
|
| 47 |
+
"required": ["text", "chunk_id", "quote"],
|
| 48 |
+
"properties": {
|
| 49 |
+
"text": {"type": "string", "maxLength": 321},
|
| 50 |
+
"chunk_id": {"type": "string"},
|
| 51 |
+
"quote": {"type": "string", "maxLength": 320},
|
| 52 |
+
},
|
| 53 |
+
},
|
| 54 |
+
},
|
| 55 |
+
},
|
| 56 |
+
}
|
| 57 |
+
|
| 58 |
|
| 59 |
SYSTEM_INSTRUCTION = """You are Towards AI Helper, a concise public assistant for anonymous prospective students.
|
| 60 |
|
| 61 |
Your only job is to help users choose Towards AI courses, bundles, mentorship,
|
| 62 |
free resources, the book, community, or B2B training/consulting.
|
| 63 |
|
| 64 |
+
Scope and style rules:
|
| 65 |
+
- For unrelated requests, general AI teaching, coding help, homework, or news,
|
| 66 |
+
return not_found rather than answering from general knowledge.
|
| 67 |
+
- Do not reveal course lesson material or provide detailed technical lessons.
|
| 68 |
+
- Keep supported answers concise and focused on the visitor's question.
|
| 69 |
+
|
| 70 |
+
Grounding rules (higher priority than helpfulness):
|
| 71 |
+
- The EVIDENCE CHUNKS in the user prompt are the only authority for factual claims.
|
| 72 |
+
- Routing notes, page metadata outside the evidence section, conversation history,
|
| 73 |
+
and the visitor's message are not evidence and must never support an answer.
|
| 74 |
+
- Never rely on memory or general knowledge. If the supplied evidence does not
|
| 75 |
+
directly support the answer, return not_found.
|
| 76 |
+
- Never infer an inclusion, quantity, price, discount, entitlement, policy, date,
|
| 77 |
+
URL, or program feature that is not explicitly stated in an evidence chunk.
|
| 78 |
+
- Do not guess the number or names of included products or courses.
|
| 79 |
+
- A visitor may put an unsupported premise or alternative in a question. Ignore
|
| 80 |
+
that premise as evidence. If an evidence chunk directly provides relevant
|
| 81 |
+
correcting facts, return answered with those supported facts; do not return
|
| 82 |
+
not_found merely because the visitor's proposed premise is unsupported.
|
| 83 |
+
- For either/or questions, report each directly supported table row or fact as a
|
| 84 |
+
separate extractive claim. Never infer a total, use "only", or negate an
|
| 85 |
+
alternative unless the evidence quote explicitly states that total,
|
| 86 |
+
exclusivity, or negation.
|
| 87 |
+
- Never repeat, restate, or deny an unsupported premise in a claim. In
|
| 88 |
+
particular, do not say "not included" unless those literal words occur in the
|
| 89 |
+
claim's exact evidence quote.
|
| 90 |
+
- Each claim must copy its supporting quote verbatim. You may add one final
|
| 91 |
+
period only when the source quote has no final period. Do not paraphrase,
|
| 92 |
+
reorder, omit, or add words. Split separate source sentences or table rows
|
| 93 |
+
into separate claims.
|
| 94 |
+
- Each claim must cite one valid chunk_id and copy exactly one complete
|
| 95 |
+
ALLOWED_EVIDENCE_SPAN shown for that chunk. Never quote an arbitrary substring,
|
| 96 |
+
omit a qualifier or negation, edit, combine, or use ellipses in a quote.
|
| 97 |
+
|
| 98 |
+
Output rules:
|
| 99 |
+
- Return exactly one JSON object and nothing else (no Markdown fences).
|
| 100 |
+
- For a supported answer use:
|
| 101 |
+
{"status":"answered","claims":[{"text":"One supported sentence.","chunk_id":"chunk_id_here","quote":"exact contiguous source quote"}]}
|
| 102 |
+
- If support is absent, ambiguous, conflicting, or insufficient, use exactly:
|
| 103 |
+
{"status":"not_found","claims":[]}
|
| 104 |
+
- Do not include an answer field, preamble, uncited sentence, or extra key.
|
| 105 |
+
- Keep supported answers to at most 6 short claims. Include URLs only when the
|
| 106 |
+
exact URL occurs in the quoted evidence.
|
| 107 |
"""
|
| 108 |
|
| 109 |
|
| 110 |
@dataclass(frozen=True)
|
| 111 |
class LLMResult:
|
| 112 |
+
"""Raw provider result.
|
| 113 |
+
|
| 114 |
+
``answer`` contains the provider's JSON text until it has passed
|
| 115 |
+
:func:`validate_grounded_result`. Callers must not send this field directly
|
| 116 |
+
to a visitor.
|
| 117 |
+
"""
|
| 118 |
+
|
| 119 |
answer: str
|
| 120 |
usage: dict[str, Any] = field(default_factory=dict)
|
| 121 |
latency_ms: int = 0
|
| 122 |
|
| 123 |
|
| 124 |
+
@dataclass(frozen=True)
|
| 125 |
+
class EvidenceSpan:
|
| 126 |
+
span_id: str
|
| 127 |
+
text: str
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
@dataclass(frozen=True)
|
| 131 |
+
class EvidenceChunk:
|
| 132 |
+
chunk_id: str
|
| 133 |
+
title: str
|
| 134 |
+
url: str
|
| 135 |
+
kind: str
|
| 136 |
+
headings: tuple[str, ...]
|
| 137 |
+
text: str
|
| 138 |
+
offer_id: str = ""
|
| 139 |
+
entity_id: str = ""
|
| 140 |
+
evidence_spans: tuple[EvidenceSpan, ...] = ()
|
| 141 |
+
|
| 142 |
+
@property
|
| 143 |
+
def evidence_text(self) -> str:
|
| 144 |
+
spans = "\n".join(
|
| 145 |
+
f'<SPAN span_id="{span.span_id}">{span.text}</SPAN>'
|
| 146 |
+
for span in self.evidence_spans
|
| 147 |
+
)
|
| 148 |
+
return "\n".join(
|
| 149 |
+
[
|
| 150 |
+
f"Title: {self.title}",
|
| 151 |
+
f"Kind: {self.kind}",
|
| 152 |
+
f"Offer ID: {self.offer_id or '(none)'}",
|
| 153 |
+
f"Entity ID: {self.entity_id or '(none)'}",
|
| 154 |
+
f"URL: {self.url}",
|
| 155 |
+
f"Headings: {', '.join(self.headings)}",
|
| 156 |
+
"Allowed evidence spans (quote exactly one complete span):",
|
| 157 |
+
spans or "(none)",
|
| 158 |
+
]
|
| 159 |
+
)
|
| 160 |
+
|
| 161 |
+
|
| 162 |
+
@dataclass(frozen=True)
|
| 163 |
+
class GroundedClaim:
|
| 164 |
+
text: str
|
| 165 |
+
chunk_id: str
|
| 166 |
+
quote: str
|
| 167 |
+
|
| 168 |
+
|
| 169 |
+
GroundingStatus = Literal["answered", "not_found", "validation_failure"]
|
| 170 |
+
|
| 171 |
+
|
| 172 |
+
@dataclass(frozen=True)
|
| 173 |
+
class GroundingResult:
|
| 174 |
+
"""Safe result returned by the deterministic grounding boundary."""
|
| 175 |
+
|
| 176 |
+
valid: bool
|
| 177 |
+
status: GroundingStatus
|
| 178 |
+
answer: str = ""
|
| 179 |
+
claims: tuple[GroundedClaim, ...] = ()
|
| 180 |
+
cited_chunks: tuple[EvidenceChunk, ...] = ()
|
| 181 |
+
usage: dict[str, Any] = field(default_factory=dict)
|
| 182 |
+
latency_ms: int = 0
|
| 183 |
+
validation_error: str = ""
|
| 184 |
+
|
| 185 |
+
@property
|
| 186 |
+
def is_answered(self) -> bool:
|
| 187 |
+
return self.valid and self.status == "answered"
|
| 188 |
+
|
| 189 |
+
@property
|
| 190 |
+
def cited_chunk_ids(self) -> tuple[str, ...]:
|
| 191 |
+
return tuple(chunk.chunk_id for chunk in self.cited_chunks)
|
| 192 |
+
|
| 193 |
+
|
| 194 |
+
_SAFE_CHUNK_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:-]{0,127}$")
|
| 195 |
+
_WHITESPACE_RE = re.compile(r"\s+")
|
| 196 |
+
_TOKEN_RE = re.compile(r"[a-z0-9]+(?:[-'][a-z0-9]+)*", re.IGNORECASE)
|
| 197 |
+
_SENTENCE_BOUNDARY_RE = re.compile(r"[.!?][\"')\]]*\s+(?=[A-Z0-9])")
|
| 198 |
+
_ATOMIC_SENTENCE_SPLIT_RE = re.compile(r"(?<=[.!?])\s+(?=[\"'(\[]*[A-Z0-9])")
|
| 199 |
+
_URL_RE = re.compile(r"https?://[^\s<>\"']+", re.IGNORECASE)
|
| 200 |
+
_PERCENT_RE = re.compile(r"(?<![\w.])\d+(?:[.,]\d+)*\s*%")
|
| 201 |
+
_CURRENCY_RE = re.compile(
|
| 202 |
+
r"(?<!\w)(?:(?:US|CA|AU)?[$€£¥]\s*\d+(?:[.,]\d+)*"
|
| 203 |
+
r"|\d+(?:[.,]\d+)*\s*(?:USD|CAD|AUD|EUR|GBP|JPY))(?!\w)",
|
| 204 |
+
re.IGNORECASE,
|
| 205 |
+
)
|
| 206 |
+
_NUMBER_RE = re.compile(r"(?<![\w.])\d+(?:[.,]\d+)*(?![\w.])")
|
| 207 |
+
_POLARITY_RE = re.compile(
|
| 208 |
+
r"\b(?:no|not|never|none|without|cannot|can't|couldn't|doesn't|don't|"
|
| 209 |
+
r"isn't|aren't|won't)\b",
|
| 210 |
+
re.IGNORECASE,
|
| 211 |
+
)
|
| 212 |
+
_UNSAFE_COMPARISON_EVIDENCE_RE = re.compile(
|
| 213 |
+
r"\b(?:a single mentor|senior ai consultant|chatgpt\s*&\s*claude|"
|
| 214 |
+
r"discord\s*&\s*reddit)\b",
|
| 215 |
+
re.IGNORECASE,
|
| 216 |
+
)
|
| 217 |
+
_NUMBER_WORDS = frozenset(
|
| 218 |
+
[
|
| 219 |
+
"zero",
|
| 220 |
+
"one",
|
| 221 |
+
"two",
|
| 222 |
+
"three",
|
| 223 |
+
"four",
|
| 224 |
+
"five",
|
| 225 |
+
"six",
|
| 226 |
+
"seven",
|
| 227 |
+
"eight",
|
| 228 |
+
"nine",
|
| 229 |
+
"ten",
|
| 230 |
+
"eleven",
|
| 231 |
+
"twelve",
|
| 232 |
+
"thirteen",
|
| 233 |
+
"fourteen",
|
| 234 |
+
"fifteen",
|
| 235 |
+
"sixteen",
|
| 236 |
+
"seventeen",
|
| 237 |
+
"eighteen",
|
| 238 |
+
"nineteen",
|
| 239 |
+
"twenty",
|
| 240 |
+
"thirty",
|
| 241 |
+
"forty",
|
| 242 |
+
"fifty",
|
| 243 |
+
"sixty",
|
| 244 |
+
"seventy",
|
| 245 |
+
"eighty",
|
| 246 |
+
"ninety",
|
| 247 |
+
"hundred",
|
| 248 |
+
"thousand",
|
| 249 |
+
"million",
|
| 250 |
+
"billion",
|
| 251 |
+
]
|
| 252 |
+
)
|
| 253 |
+
|
| 254 |
+
_QUERY_RELEVANCE_STOP = frozenset(
|
| 255 |
+
{
|
| 256 |
+
"a",
|
| 257 |
+
"about",
|
| 258 |
+
"all",
|
| 259 |
+
"am",
|
| 260 |
+
"an",
|
| 261 |
+
"and",
|
| 262 |
+
"any",
|
| 263 |
+
"are",
|
| 264 |
+
"as",
|
| 265 |
+
"at",
|
| 266 |
+
"be",
|
| 267 |
+
"can",
|
| 268 |
+
"come",
|
| 269 |
+
"comes",
|
| 270 |
+
"course",
|
| 271 |
+
"courses",
|
| 272 |
+
"do",
|
| 273 |
+
"does",
|
| 274 |
+
"every",
|
| 275 |
+
"each",
|
| 276 |
+
"feature",
|
| 277 |
+
"features",
|
| 278 |
+
"for",
|
| 279 |
+
"from",
|
| 280 |
+
"get",
|
| 281 |
+
"give",
|
| 282 |
+
"has",
|
| 283 |
+
"have",
|
| 284 |
+
"having",
|
| 285 |
+
"how",
|
| 286 |
+
"i",
|
| 287 |
+
"if",
|
| 288 |
+
"in",
|
| 289 |
+
"include",
|
| 290 |
+
"included",
|
| 291 |
+
"includes",
|
| 292 |
+
"is",
|
| 293 |
+
"it",
|
| 294 |
+
"me",
|
| 295 |
+
"many",
|
| 296 |
+
"my",
|
| 297 |
+
"of",
|
| 298 |
+
"on",
|
| 299 |
+
"or",
|
| 300 |
+
"our",
|
| 301 |
+
"part",
|
| 302 |
+
"per",
|
| 303 |
+
"plan",
|
| 304 |
+
"please",
|
| 305 |
+
"program",
|
| 306 |
+
"provide",
|
| 307 |
+
"really",
|
| 308 |
+
"that",
|
| 309 |
+
"the",
|
| 310 |
+
"their",
|
| 311 |
+
"there",
|
| 312 |
+
"this",
|
| 313 |
+
"to",
|
| 314 |
+
"towards",
|
| 315 |
+
"true",
|
| 316 |
+
"us",
|
| 317 |
+
"we",
|
| 318 |
+
"what",
|
| 319 |
+
"which",
|
| 320 |
+
"with",
|
| 321 |
+
"long",
|
| 322 |
+
"you",
|
| 323 |
+
"your",
|
| 324 |
+
}
|
| 325 |
+
)
|
| 326 |
+
|
| 327 |
+
_FACT_QUERY_TERMS = frozenset(
|
| 328 |
+
{
|
| 329 |
+
"access",
|
| 330 |
+
"annual",
|
| 331 |
+
"annually",
|
| 332 |
+
"cancel",
|
| 333 |
+
"cancelled",
|
| 334 |
+
"cancelling",
|
| 335 |
+
"certificate",
|
| 336 |
+
"certification",
|
| 337 |
+
"cost",
|
| 338 |
+
"coupon",
|
| 339 |
+
"discount",
|
| 340 |
+
"duration",
|
| 341 |
+
"forever",
|
| 342 |
+
"guarantee",
|
| 343 |
+
"guaranteed",
|
| 344 |
+
"hour",
|
| 345 |
+
"hours",
|
| 346 |
+
"keep",
|
| 347 |
+
"lesson",
|
| 348 |
+
"lessons",
|
| 349 |
+
"lifetime",
|
| 350 |
+
"money-back",
|
| 351 |
+
"month",
|
| 352 |
+
"month-by-month",
|
| 353 |
+
"month-to-month",
|
| 354 |
+
"monthly",
|
| 355 |
+
"page",
|
| 356 |
+
"pages",
|
| 357 |
+
"price",
|
| 358 |
+
"product",
|
| 359 |
+
"products",
|
| 360 |
+
"promo",
|
| 361 |
+
"prerequisite",
|
| 362 |
+
"prerequisites",
|
| 363 |
+
"refund",
|
| 364 |
+
"retain",
|
| 365 |
+
"year",
|
| 366 |
+
"yearly",
|
| 367 |
+
}
|
| 368 |
+
)
|
| 369 |
+
|
| 370 |
+
|
| 371 |
+
def _normalize_text(value: str) -> str:
|
| 372 |
+
normalized = unicodedata.normalize("NFKC", value).replace("’", "'")
|
| 373 |
+
return _WHITESPACE_RE.sub(" ", normalized).strip()
|
| 374 |
+
|
| 375 |
+
|
| 376 |
+
def chunk_id_for_page(page: dict[str, Any]) -> str:
|
| 377 |
+
"""Return a deterministic, retrieval-order-independent evidence ID."""
|
| 378 |
+
|
| 379 |
+
explicit = str(page.get("chunk_id", "")).strip()
|
| 380 |
+
if explicit and _SAFE_CHUNK_ID_RE.fullmatch(explicit):
|
| 381 |
+
return explicit
|
| 382 |
+
|
| 383 |
+
identity = json.dumps(
|
| 384 |
+
{
|
| 385 |
+
"url": str(page.get("url", "")),
|
| 386 |
+
"title": str(page.get("title", "")),
|
| 387 |
+
"kind": str(page.get("kind", "")),
|
| 388 |
+
"chunk_index": page.get("chunk_index"),
|
| 389 |
+
"text": _normalize_text(str(page.get("text", ""))),
|
| 390 |
+
},
|
| 391 |
+
ensure_ascii=False,
|
| 392 |
+
sort_keys=True,
|
| 393 |
+
separators=(",", ":"),
|
| 394 |
+
)
|
| 395 |
+
digest = hashlib.sha256(identity.encode("utf-8")).hexdigest()[:16]
|
| 396 |
+
return f"chunk_{digest}"
|
| 397 |
+
|
| 398 |
+
|
| 399 |
+
def _fallback_span_texts(text: str) -> list[str]:
|
| 400 |
+
"""Derive conservative atomic spans for tests and legacy in-memory inputs."""
|
| 401 |
+
|
| 402 |
+
normalized = _normalize_text(text)
|
| 403 |
+
if not normalized:
|
| 404 |
+
return []
|
| 405 |
+
candidates = _ATOMIC_SENTENCE_SPLIT_RE.split(normalized)
|
| 406 |
+
result: list[str] = []
|
| 407 |
+
for candidate in candidates:
|
| 408 |
+
span = _normalize_text(candidate)
|
| 409 |
+
if 2 <= len(_TOKEN_RE.findall(span)) and len(span) <= MAX_EVIDENCE_QUOTE_CHARS:
|
| 410 |
+
result.append(span)
|
| 411 |
+
return list(dict.fromkeys(result))
|
| 412 |
+
|
| 413 |
+
|
| 414 |
+
def _evidence_spans_from_page(
|
| 415 |
+
page: dict[str, Any], chunk_id: str, text: str
|
| 416 |
+
) -> tuple[EvidenceSpan, ...]:
|
| 417 |
+
normalized_text = _normalize_text(text)
|
| 418 |
+
raw_spans = page.get("evidence_spans")
|
| 419 |
+
candidates: list[tuple[str, str]] = []
|
| 420 |
+
if isinstance(raw_spans, list):
|
| 421 |
+
for index, raw_span in enumerate(raw_spans):
|
| 422 |
+
if not isinstance(raw_span, dict):
|
| 423 |
+
continue
|
| 424 |
+
span_text = _normalize_text(str(raw_span.get("text", "")))
|
| 425 |
+
span_id = str(raw_span.get("span_id", "")).strip()
|
| 426 |
+
if not span_id:
|
| 427 |
+
span_id = f"{chunk_id}:span-{index}"
|
| 428 |
+
candidates.append((span_id, span_text))
|
| 429 |
+
else:
|
| 430 |
+
candidates = [
|
| 431 |
+
(f"{chunk_id}:span-{index}", span_text)
|
| 432 |
+
for index, span_text in enumerate(_fallback_span_texts(text))
|
| 433 |
+
]
|
| 434 |
+
|
| 435 |
+
result: list[EvidenceSpan] = []
|
| 436 |
+
seen: set[tuple[str, str]] = set()
|
| 437 |
+
for span_id, span_text in candidates:
|
| 438 |
+
key = (span_id, span_text)
|
| 439 |
+
if (
|
| 440 |
+
key in seen
|
| 441 |
+
or not _SAFE_CHUNK_ID_RE.fullmatch(span_id)
|
| 442 |
+
or not span_text
|
| 443 |
+
or len(span_text) > MAX_EVIDENCE_QUOTE_CHARS
|
| 444 |
+
or span_text not in normalized_text
|
| 445 |
+
):
|
| 446 |
+
continue
|
| 447 |
+
seen.add(key)
|
| 448 |
+
result.append(EvidenceSpan(span_id=span_id, text=span_text))
|
| 449 |
+
return tuple(result)
|
| 450 |
+
|
| 451 |
+
|
| 452 |
+
def _chunk_from_page(page: dict[str, Any]) -> EvidenceChunk:
|
| 453 |
+
chunk_id = chunk_id_for_page(page)
|
| 454 |
+
text = str(page.get("text", ""))[:MAX_CHUNK_TEXT_CHARS]
|
| 455 |
+
return EvidenceChunk(
|
| 456 |
+
chunk_id=chunk_id,
|
| 457 |
+
title=str(page.get("title", "")),
|
| 458 |
+
url=str(page.get("url", "")),
|
| 459 |
+
kind=str(page.get("kind", "")),
|
| 460 |
+
headings=tuple(str(item) for item in page.get("headings", [])[:12]),
|
| 461 |
+
text=text,
|
| 462 |
+
offer_id=str(page.get("offer_id", "")),
|
| 463 |
+
entity_id=str(page.get("entity_id", "")),
|
| 464 |
+
evidence_spans=_evidence_spans_from_page(page, chunk_id, text),
|
| 465 |
+
)
|
| 466 |
+
|
| 467 |
+
|
| 468 |
+
def _render_chunk(chunk: EvidenceChunk) -> str:
|
| 469 |
+
return "\n".join(
|
| 470 |
+
[
|
| 471 |
+
f'<EVIDENCE_CHUNK chunk_id="{chunk.chunk_id}">',
|
| 472 |
+
chunk.evidence_text,
|
| 473 |
+
"</EVIDENCE_CHUNK>",
|
| 474 |
+
]
|
| 475 |
+
)
|
| 476 |
+
|
| 477 |
+
|
| 478 |
+
def _routing_notes_section() -> str:
|
| 479 |
+
return (
|
| 480 |
+
"<NON_EVIDENCE_ROUTING_NOTES>\n"
|
| 481 |
+
"Routing was completed deterministically before generation. This section\n"
|
| 482 |
+
"contains no factual evidence, must not support a claim, and must not be cited.\n"
|
| 483 |
+
"</NON_EVIDENCE_ROUTING_NOTES>"
|
| 484 |
+
)
|
| 485 |
+
|
| 486 |
+
|
| 487 |
+
def evidence_chunks(
|
| 488 |
+
selected_pages: list[dict[str, Any]], *, max_chars: int = MAX_CONTEXT_CHARS
|
| 489 |
+
) -> tuple[EvidenceChunk, ...]:
|
| 490 |
+
"""Return exactly the source chunks that fit in the model evidence block."""
|
| 491 |
+
|
| 492 |
+
routing_section = _routing_notes_section()
|
| 493 |
+
used_chars = len(routing_section)
|
| 494 |
+
chunks: list[EvidenceChunk] = []
|
| 495 |
+
seen_ids: set[str] = set()
|
| 496 |
+
|
| 497 |
+
for page in selected_pages:
|
| 498 |
if str(page.get("url", "")).startswith("internal://"):
|
| 499 |
continue
|
| 500 |
+
chunk = _chunk_from_page(page)
|
| 501 |
+
if chunk.chunk_id in seen_ids:
|
| 502 |
+
continue
|
| 503 |
+
rendered_length = len(_render_chunk(chunk)) + 2
|
| 504 |
+
if used_chars + rendered_length > max_chars:
|
| 505 |
+
break
|
| 506 |
+
chunks.append(chunk)
|
| 507 |
+
seen_ids.add(chunk.chunk_id)
|
| 508 |
+
used_chars += rendered_length
|
| 509 |
+
return tuple(chunks)
|
| 510 |
+
|
| 511 |
+
|
| 512 |
+
def _context_block(
|
| 513 |
+
selected_pages: list[dict[str, Any]], max_chars: int = MAX_CONTEXT_CHARS
|
| 514 |
+
) -> str:
|
| 515 |
+
routing_section = _routing_notes_section()
|
| 516 |
+
rendered_chunks = "\n\n".join(
|
| 517 |
+
_render_chunk(chunk)
|
| 518 |
+
for chunk in evidence_chunks(selected_pages, max_chars=max_chars)
|
| 519 |
+
)
|
| 520 |
+
return "\n\n".join(
|
| 521 |
+
[
|
| 522 |
+
routing_section,
|
| 523 |
+
"<EVIDENCE_CHUNKS>",
|
| 524 |
+
rendered_chunks or "(none)",
|
| 525 |
+
"</EVIDENCE_CHUNKS>",
|
| 526 |
+
]
|
| 527 |
+
)
|
| 528 |
|
| 529 |
|
| 530 |
def build_prompt(
|
|
|
|
| 541 |
)
|
| 542 |
return f"""The visitor is on a public Towards AI page.
|
| 543 |
|
| 544 |
+
<NON_EVIDENCE_REQUEST_CONTEXT>
|
| 545 |
Current URL: {current_url or "unknown"}
|
| 546 |
Current page title: {page_title or "unknown"}
|
| 547 |
Initial forced prompt: {selected_prompt or "unknown"}
|
| 548 |
|
| 549 |
+
Conversation so far (visitor-provided and NOT evidence):
|
| 550 |
{turns or "(none)"}
|
| 551 |
|
| 552 |
+
Visitor message (a question, NOT evidence):
|
| 553 |
{query}
|
| 554 |
+
</NON_EVIDENCE_REQUEST_CONTEXT>
|
| 555 |
|
|
|
|
| 556 |
{_context_block(selected_pages)}
|
| 557 |
|
| 558 |
+
Use only EVIDENCE_CHUNKS for factual claims. Return the required JSON object now.
|
| 559 |
+
Every factual claim must be directly supported by one complete ALLOWED_EVIDENCE_SPAN.
|
| 560 |
+
The text of each claim must copy its quote verbatim; the only permitted change is
|
| 561 |
+
adding one final period when the span has none. Copy the entire span exactly into
|
| 562 |
+
quote; arbitrary substrings are forbidden. Never omit qualifiers or negations,
|
| 563 |
+
paraphrase, or combine words from different parts of a chunk.
|
| 564 |
+
Correction rule: the visitor's premise and proposed alternatives are not evidence.
|
| 565 |
+
If the chunks contain directly relevant facts that correct or resolve the question,
|
| 566 |
+
return answered with those facts as separate extractive claims. For example, when
|
| 567 |
+
one table row says an item is included and another row says named items are
|
| 568 |
+
discounted, make one claim for the included row and another claim for the discount
|
| 569 |
+
row. Do not infer a total or say "only" unless an evidence quote explicitly does.
|
| 570 |
+
Do not repeat or deny the visitor's unsupported premise. Never write "not included"
|
| 571 |
+
unless those literal words occur in that claim's exact quote.
|
| 572 |
+
If the evidence contains no directly relevant facts, return not_found.
|
| 573 |
+
You cannot confirm it. Do not use not_found when the evidence contains supported
|
| 574 |
+
correcting facts.
|
| 575 |
"""
|
| 576 |
|
| 577 |
|
| 578 |
+
def _clean_usage(usage: dict[str, Any]) -> dict[str, Any]:
|
| 579 |
+
return {key: value for key, value in usage.items() if value is not None}
|
| 580 |
+
|
| 581 |
+
|
| 582 |
+
def _text_from_chat_content(content: Any) -> str:
|
| 583 |
+
if isinstance(content, str):
|
| 584 |
+
return content
|
| 585 |
+
if isinstance(content, list):
|
| 586 |
+
parts: list[str] = []
|
| 587 |
+
for item in content:
|
| 588 |
+
if isinstance(item, dict):
|
| 589 |
+
parts.append(str(item.get("text", "")))
|
| 590 |
+
else:
|
| 591 |
+
parts.append(str(item))
|
| 592 |
+
return "".join(parts)
|
| 593 |
+
return str(content or "")
|
| 594 |
+
|
| 595 |
+
|
| 596 |
+
def _deepseek_headers() -> dict[str, str]:
|
| 597 |
+
return {
|
| 598 |
+
"Authorization": f"Bearer {settings.deepseek_api_key}",
|
| 599 |
+
"Content-Type": "application/json",
|
| 600 |
+
}
|
| 601 |
+
|
| 602 |
+
|
| 603 |
+
def _generate_deepseek_answer(prompt: str) -> LLMResult:
|
| 604 |
+
if not settings.deepseek_api_key:
|
| 605 |
+
raise RuntimeError("DEEPSEEK_API_KEY is not configured")
|
| 606 |
+
|
| 607 |
+
response = requests.post(
|
| 608 |
+
f"{settings.deepseek_base_url}/chat/completions",
|
| 609 |
+
headers=_deepseek_headers(),
|
| 610 |
+
json={
|
| 611 |
+
"model": settings.primary_model_name,
|
| 612 |
+
"messages": [
|
| 613 |
+
{"role": "system", "content": SYSTEM_INSTRUCTION},
|
| 614 |
+
{"role": "user", "content": prompt},
|
| 615 |
+
],
|
| 616 |
+
"thinking": {"type": settings.deepseek_thinking_type},
|
| 617 |
+
"response_format": {"type": "json_object"},
|
| 618 |
+
"temperature": DEFAULT_TEMPERATURE,
|
| 619 |
+
"max_tokens": settings.max_output_tokens,
|
| 620 |
+
"stream": False,
|
| 621 |
+
},
|
| 622 |
+
timeout=settings.llm_request_timeout_seconds,
|
| 623 |
+
)
|
| 624 |
+
if response.status_code >= 400:
|
| 625 |
+
details = response.text.strip()[:500]
|
| 626 |
+
raise RuntimeError(
|
| 627 |
+
f"DeepSeek request failed with HTTP {response.status_code}: {details}"
|
| 628 |
+
)
|
| 629 |
+
|
| 630 |
+
payload = response.json()
|
| 631 |
+
choices = payload.get("choices") or []
|
| 632 |
+
if not choices:
|
| 633 |
+
raise RuntimeError("DeepSeek response did not include any choices")
|
| 634 |
+
|
| 635 |
+
message = choices[0].get("message") or {}
|
| 636 |
+
answer = _text_from_chat_content(message.get("content")).strip()
|
| 637 |
+
raw_usage = payload.get("usage") or {}
|
| 638 |
+
usage = _clean_usage(
|
| 639 |
+
{
|
| 640 |
+
"input_tokens": raw_usage.get("prompt_tokens")
|
| 641 |
+
or raw_usage.get("input_tokens"),
|
| 642 |
+
"output_tokens": raw_usage.get("completion_tokens")
|
| 643 |
+
or raw_usage.get("output_tokens"),
|
| 644 |
+
"total_tokens": raw_usage.get("total_tokens"),
|
| 645 |
+
"provider": DEEPSEEK_PROVIDER,
|
| 646 |
+
"model": settings.primary_model_name,
|
| 647 |
+
}
|
| 648 |
+
)
|
| 649 |
+
return LLMResult(answer=answer, usage=usage)
|
| 650 |
+
|
| 651 |
+
|
| 652 |
+
def _generate_gemini_answer(prompt: str, *, fallback_from: str = "") -> LLMResult:
|
| 653 |
if not settings.gemini_api_key:
|
| 654 |
raise RuntimeError("GEMINI_API_KEY is not configured")
|
| 655 |
|
|
|
|
| 656 |
client = genai.Client(api_key=settings.gemini_api_key)
|
| 657 |
response = client.models.generate_content(
|
| 658 |
+
model=settings.fallback_model_name,
|
| 659 |
contents=prompt,
|
| 660 |
config={
|
| 661 |
"system_instruction": SYSTEM_INSTRUCTION,
|
| 662 |
+
"temperature": DEFAULT_TEMPERATURE,
|
| 663 |
"max_output_tokens": settings.max_output_tokens,
|
| 664 |
+
"response_mime_type": "application/json",
|
| 665 |
+
"response_json_schema": GROUNDING_RESPONSE_SCHEMA,
|
| 666 |
},
|
| 667 |
)
|
| 668 |
usage_metadata = getattr(response, "usage_metadata", None)
|
|
|
|
| 673 |
"output_tokens": getattr(usage_metadata, "candidates_token_count", None),
|
| 674 |
"total_tokens": getattr(usage_metadata, "total_token_count", None),
|
| 675 |
}
|
| 676 |
+
usage.update(
|
| 677 |
+
{
|
| 678 |
+
"provider": GEMINI_PROVIDER,
|
| 679 |
+
"model": settings.fallback_model_name,
|
| 680 |
+
}
|
| 681 |
+
)
|
| 682 |
+
if fallback_from:
|
| 683 |
+
usage["fallback_from"] = fallback_from
|
| 684 |
return LLMResult(
|
| 685 |
answer=(getattr(response, "text", "") or "").strip(),
|
| 686 |
+
usage=_clean_usage(usage),
|
| 687 |
+
)
|
| 688 |
+
|
| 689 |
+
|
| 690 |
+
def _with_latency(result: LLMResult, started: float) -> LLMResult:
|
| 691 |
+
return replace(result, latency_ms=int((time.monotonic() - started) * 1000))
|
| 692 |
+
|
| 693 |
+
|
| 694 |
+
def generate_answer(prompt: str) -> LLMResult:
|
| 695 |
+
"""Generate the raw structured provider response, retaining provider fallback."""
|
| 696 |
+
|
| 697 |
+
started = time.monotonic()
|
| 698 |
+
primary_error: Exception | None = None
|
| 699 |
+
|
| 700 |
+
if settings.deepseek_api_key:
|
| 701 |
+
try:
|
| 702 |
+
result = _generate_deepseek_answer(prompt)
|
| 703 |
+
return _with_latency(result, started)
|
| 704 |
+
except Exception as exc:
|
| 705 |
+
primary_error = exc
|
| 706 |
+
logger.warning(
|
| 707 |
+
"DeepSeek generation failed; falling back to Gemini.",
|
| 708 |
+
exc_info=True,
|
| 709 |
+
)
|
| 710 |
+
|
| 711 |
+
try:
|
| 712 |
+
result = _generate_gemini_answer(
|
| 713 |
+
prompt,
|
| 714 |
+
fallback_from=DEEPSEEK_PROVIDER if primary_error else "",
|
| 715 |
+
)
|
| 716 |
+
return _with_latency(result, started)
|
| 717 |
+
except Exception as fallback_error:
|
| 718 |
+
if primary_error is not None:
|
| 719 |
+
raise RuntimeError(
|
| 720 |
+
"DeepSeek primary and Gemini fallback generation both failed"
|
| 721 |
+
) from fallback_error
|
| 722 |
+
raise
|
| 723 |
+
|
| 724 |
+
|
| 725 |
+
class _DuplicateKeyError(ValueError):
|
| 726 |
+
pass
|
| 727 |
+
|
| 728 |
+
|
| 729 |
+
def _reject_duplicate_keys(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
| 730 |
+
result: dict[str, Any] = {}
|
| 731 |
+
for key, value in pairs:
|
| 732 |
+
if key in result:
|
| 733 |
+
raise _DuplicateKeyError(f"duplicate JSON key: {key}")
|
| 734 |
+
result[key] = value
|
| 735 |
+
return result
|
| 736 |
+
|
| 737 |
+
|
| 738 |
+
def _reject_json_constant(value: str) -> None:
|
| 739 |
+
raise ValueError(f"non-standard JSON constant: {value}")
|
| 740 |
+
|
| 741 |
+
|
| 742 |
+
def _validation_failure(raw: LLMResult, message: str) -> GroundingResult:
|
| 743 |
+
return GroundingResult(
|
| 744 |
+
valid=False,
|
| 745 |
+
status="validation_failure",
|
| 746 |
+
usage=raw.usage,
|
| 747 |
+
latency_ms=raw.latency_ms,
|
| 748 |
+
validation_error=message,
|
| 749 |
+
)
|
| 750 |
+
|
| 751 |
+
|
| 752 |
+
def _trim_url(value: str) -> str:
|
| 753 |
+
return value.rstrip(".,;:!?)]}").casefold()
|
| 754 |
+
|
| 755 |
+
|
| 756 |
+
def _critical_facts(value: str) -> dict[str, set[str]]:
|
| 757 |
+
normalized = _normalize_text(value)
|
| 758 |
+
word_tokens = {token.casefold() for token in _TOKEN_RE.findall(normalized)}
|
| 759 |
+
return {
|
| 760 |
+
"URLs": {_trim_url(item) for item in _URL_RE.findall(normalized)},
|
| 761 |
+
"percentages": {
|
| 762 |
+
_normalize_text(item).casefold() for item in _PERCENT_RE.findall(normalized)
|
| 763 |
+
},
|
| 764 |
+
"currency amounts": {
|
| 765 |
+
_normalize_text(item).casefold()
|
| 766 |
+
for item in _CURRENCY_RE.findall(normalized)
|
| 767 |
+
},
|
| 768 |
+
"numbers": {
|
| 769 |
+
_normalize_text(item).casefold() for item in _NUMBER_RE.findall(normalized)
|
| 770 |
+
},
|
| 771 |
+
"number words": word_tokens & _NUMBER_WORDS,
|
| 772 |
+
}
|
| 773 |
+
|
| 774 |
+
|
| 775 |
+
def _claim_is_one_sentence(text: str) -> bool:
|
| 776 |
+
if not text or re.search(r"[.!?][\"')\]]*$", text) is None:
|
| 777 |
+
return False
|
| 778 |
+
return _SENTENCE_BOUNDARY_RE.search(text) is None
|
| 779 |
+
|
| 780 |
+
|
| 781 |
+
def _validate_claim_reference(
|
| 782 |
+
claim: GroundedClaim, chunk_by_id: dict[str, EvidenceChunk]
|
| 783 |
+
) -> str:
|
| 784 |
+
chunk = chunk_by_id.get(claim.chunk_id)
|
| 785 |
+
if chunk is None:
|
| 786 |
+
return f"unknown chunk_id: {claim.chunk_id}"
|
| 787 |
+
|
| 788 |
+
normalized_quote = _normalize_text(claim.quote)
|
| 789 |
+
if not normalized_quote:
|
| 790 |
+
return "evidence quote must not be empty"
|
| 791 |
+
if len(normalized_quote) > MAX_EVIDENCE_QUOTE_CHARS:
|
| 792 |
+
return f"evidence quote exceeds {MAX_EVIDENCE_QUOTE_CHARS} characters"
|
| 793 |
+
normalized_chunk = _normalize_text(chunk.text)
|
| 794 |
+
if normalized_quote not in normalized_chunk:
|
| 795 |
+
return "evidence span is not present in its chunk text"
|
| 796 |
+
allowed_spans = {
|
| 797 |
+
_normalize_text(span.text) for span in chunk.evidence_spans if span.text
|
| 798 |
+
}
|
| 799 |
+
if normalized_quote not in allowed_spans:
|
| 800 |
+
return "evidence quote must equal one complete server-defined evidence span"
|
| 801 |
+
return ""
|
| 802 |
+
|
| 803 |
+
|
| 804 |
+
def _validate_claim_text(claim: GroundedClaim) -> str:
|
| 805 |
+
normalized_claim = _normalize_text(claim.text)
|
| 806 |
+
normalized_quote = _normalize_text(claim.quote)
|
| 807 |
+
claim_facts = _critical_facts(claim.text)
|
| 808 |
+
evidence_facts = _critical_facts(claim.quote)
|
| 809 |
+
for label, facts in claim_facts.items():
|
| 810 |
+
unsupported = facts - evidence_facts[label]
|
| 811 |
+
if unsupported:
|
| 812 |
+
values = ", ".join(sorted(unsupported))
|
| 813 |
+
return f"claim contains unsupported {label}: {values}"
|
| 814 |
+
|
| 815 |
+
claim_polarity = {
|
| 816 |
+
item.casefold() for item in _POLARITY_RE.findall(_normalize_text(claim.text))
|
| 817 |
+
}
|
| 818 |
+
quote_polarity = {
|
| 819 |
+
item.casefold() for item in _POLARITY_RE.findall(_normalize_text(claim.quote))
|
| 820 |
+
}
|
| 821 |
+
if claim_polarity != quote_polarity:
|
| 822 |
+
return "claim changes or omits evidence negation"
|
| 823 |
+
|
| 824 |
+
allowed_claim_texts = {normalized_quote}
|
| 825 |
+
if not re.search(r"[.!?][\"')\]]*$", normalized_quote):
|
| 826 |
+
allowed_claim_texts.add(f"{normalized_quote}.")
|
| 827 |
+
if normalized_claim not in allowed_claim_texts:
|
| 828 |
+
return "claim must copy its evidence quote verbatim"
|
| 829 |
+
|
| 830 |
+
canonical_text = normalized_quote
|
| 831 |
+
if not re.search(r"[.!?][\"')\]]*$", canonical_text):
|
| 832 |
+
canonical_text = f"{canonical_text}."
|
| 833 |
+
if not _claim_is_one_sentence(canonical_text):
|
| 834 |
+
return "each claim must contain exactly one complete sentence"
|
| 835 |
+
|
| 836 |
+
return ""
|
| 837 |
+
|
| 838 |
+
|
| 839 |
+
def _specific_query_terms(
|
| 840 |
+
query: str, target_offer_ids: frozenset[str]
|
| 841 |
+
) -> frozenset[str]:
|
| 842 |
+
alias_terms = offer_alias_tokens(target_offer_ids)
|
| 843 |
+
result: set[str] = set()
|
| 844 |
+
for token in (item.casefold() for item in _TOKEN_RE.findall(query)):
|
| 845 |
+
if (
|
| 846 |
+
token in _QUERY_RELEVANCE_STOP
|
| 847 |
+
or token in _FACT_QUERY_TERMS
|
| 848 |
+
or token in _NUMBER_WORDS
|
| 849 |
+
or token in alias_terms
|
| 850 |
+
or any(character.isdigit() for character in token)
|
| 851 |
+
):
|
| 852 |
+
continue
|
| 853 |
+
result.add(token)
|
| 854 |
+
return frozenset(result)
|
| 855 |
+
|
| 856 |
+
|
| 857 |
+
def _term_is_present(term: str, evidence_terms: set[str]) -> bool:
|
| 858 |
+
if term in evidence_terms:
|
| 859 |
+
return True
|
| 860 |
+
if term.endswith("s") and term[:-1] in evidence_terms:
|
| 861 |
+
return True
|
| 862 |
+
return f"{term}s" in evidence_terms
|
| 863 |
+
|
| 864 |
+
|
| 865 |
+
def _validate_query_grounding(
|
| 866 |
+
query: str,
|
| 867 |
+
claims: list[GroundedClaim],
|
| 868 |
+
chunk_by_id: dict[str, EvidenceChunk],
|
| 869 |
+
target_offer_ids: frozenset[str],
|
| 870 |
+
) -> str:
|
| 871 |
+
if not query:
|
| 872 |
+
return ""
|
| 873 |
+
|
| 874 |
+
resolved_targets = target_offer_ids or offer_ids_for_query(query)
|
| 875 |
+
if resolved_targets:
|
| 876 |
+
for claim in claims:
|
| 877 |
+
chunk = chunk_by_id[claim.chunk_id]
|
| 878 |
+
if chunk.offer_id not in resolved_targets:
|
| 879 |
+
return (
|
| 880 |
+
"claim cites offer "
|
| 881 |
+
f"{chunk.offer_id or '(none)'} outside the requested offer boundary"
|
| 882 |
+
)
|
| 883 |
+
if _UNSAFE_COMPARISON_EVIDENCE_RE.search(claim.quote):
|
| 884 |
+
return "claim cites competitor comparison rather than offer evidence"
|
| 885 |
+
|
| 886 |
+
fields = requested_fact_fields(query)
|
| 887 |
+
if mixed_preview_fact_request(query):
|
| 888 |
+
return (
|
| 889 |
+
"mixed preview and paid-course facts lack separate target-qualified "
|
| 890 |
+
"evidence scopes"
|
| 891 |
+
)
|
| 892 |
+
for claim in claims:
|
| 893 |
+
chunk = chunk_by_id[claim.chunk_id]
|
| 894 |
+
if fields and not any(
|
| 895 |
+
chunk.offer_id in evidence_offer_ids_for_field(query, fact_field)
|
| 896 |
+
and evidence_span_supports_field(
|
| 897 |
+
claim.quote,
|
| 898 |
+
field=fact_field,
|
| 899 |
+
offer_id=chunk.offer_id,
|
| 900 |
+
query=query,
|
| 901 |
+
)
|
| 902 |
+
for fact_field in fields
|
| 903 |
+
):
|
| 904 |
+
return "claim lacks target-qualified evidence for the requested facts"
|
| 905 |
+
for fact_field in fields:
|
| 906 |
+
if not any(
|
| 907 |
+
chunk_by_id[claim.chunk_id].offer_id
|
| 908 |
+
in evidence_offer_ids_for_field(query, fact_field)
|
| 909 |
+
and evidence_span_supports_field(
|
| 910 |
+
claim.quote,
|
| 911 |
+
field=fact_field,
|
| 912 |
+
offer_id=chunk_by_id[claim.chunk_id].offer_id,
|
| 913 |
+
query=query,
|
| 914 |
+
)
|
| 915 |
+
for claim in claims
|
| 916 |
+
):
|
| 917 |
+
return f"answer lacks target-qualified {fact_field} evidence"
|
| 918 |
+
|
| 919 |
+
normalized_query = _normalize_text(query).casefold()
|
| 920 |
+
combined_quotes = " ".join(_normalize_text(claim.quote) for claim in claims)
|
| 921 |
+
normalized_quotes = combined_quotes.casefold()
|
| 922 |
+
if (
|
| 923 |
+
monthly_plan_intent(normalized_query)
|
| 924 |
+
and fields & {"refund", "guarantee"}
|
| 925 |
+
and not (
|
| 926 |
+
"monthly" in normalized_quotes
|
| 927 |
+
and "yearly" in normalized_quotes
|
| 928 |
+
and "money-back" in normalized_quotes
|
| 929 |
+
)
|
| 930 |
+
):
|
| 931 |
+
return "monthly guarantee answer must include the qualified yearly policy"
|
| 932 |
+
|
| 933 |
+
specific_terms = _specific_query_terms(query, resolved_targets)
|
| 934 |
+
if specific_terms:
|
| 935 |
+
evidence_terms = {
|
| 936 |
+
item.casefold() for item in _TOKEN_RE.findall(combined_quotes)
|
| 937 |
+
}
|
| 938 |
+
matched = sum(
|
| 939 |
+
_term_is_present(term, evidence_terms) for term in specific_terms
|
| 940 |
+
)
|
| 941 |
+
# Require at least two thirds of the visitor's non-generic qualifiers.
|
| 942 |
+
# Corrections involving a false number remain possible because proposed
|
| 943 |
+
# numbers and number words are deliberately excluded above.
|
| 944 |
+
if matched * 3 < len(specific_terms) * 2:
|
| 945 |
+
return "answer does not address enough of the requested qualifiers"
|
| 946 |
+
return ""
|
| 947 |
+
|
| 948 |
+
|
| 949 |
+
def validate_grounded_result(
|
| 950 |
+
raw_result: LLMResult | str,
|
| 951 |
+
selected_pages: list[dict[str, Any]],
|
| 952 |
+
*,
|
| 953 |
+
query: str = "",
|
| 954 |
+
target_offer_ids: frozenset[str] | None = None,
|
| 955 |
+
) -> GroundingResult:
|
| 956 |
+
"""Strictly validate provider JSON and return only evidence-backed text.
|
| 957 |
+
|
| 958 |
+
Any malformed schema, bad citation, unsafe evidence span, or non-verbatim
|
| 959 |
+
claim produces ``validation_failure`` with an empty ``answer``. Raw model
|
| 960 |
+
text is never copied into the safe result.
|
| 961 |
+
"""
|
| 962 |
+
|
| 963 |
+
raw = raw_result if isinstance(raw_result, LLMResult) else LLMResult(raw_result)
|
| 964 |
+
try:
|
| 965 |
+
payload = json.loads(
|
| 966 |
+
raw.answer,
|
| 967 |
+
object_pairs_hook=_reject_duplicate_keys,
|
| 968 |
+
parse_constant=_reject_json_constant,
|
| 969 |
+
)
|
| 970 |
+
except (json.JSONDecodeError, TypeError, ValueError) as exc:
|
| 971 |
+
return _validation_failure(raw, f"malformed JSON: {exc}")
|
| 972 |
+
|
| 973 |
+
if not isinstance(payload, dict) or set(payload) != {"status", "claims"}:
|
| 974 |
+
return _validation_failure(raw, "root must contain exactly status and claims")
|
| 975 |
+
status = payload["status"]
|
| 976 |
+
claims_payload = payload["claims"]
|
| 977 |
+
if status not in {"answered", "not_found"}:
|
| 978 |
+
return _validation_failure(raw, "status must be answered or not_found")
|
| 979 |
+
if not isinstance(claims_payload, list):
|
| 980 |
+
return _validation_failure(raw, "claims must be an array")
|
| 981 |
+
if len(claims_payload) > 6:
|
| 982 |
+
return _validation_failure(raw, "claims must contain at most 6 items")
|
| 983 |
+
|
| 984 |
+
if status == "not_found":
|
| 985 |
+
if claims_payload:
|
| 986 |
+
return _validation_failure(raw, "not_found must contain no claims")
|
| 987 |
+
return GroundingResult(
|
| 988 |
+
valid=True,
|
| 989 |
+
status="not_found",
|
| 990 |
+
usage=raw.usage,
|
| 991 |
+
latency_ms=raw.latency_ms,
|
| 992 |
+
)
|
| 993 |
+
if not claims_payload:
|
| 994 |
+
return _validation_failure(raw, "answered must contain at least one claim")
|
| 995 |
+
|
| 996 |
+
chunks = evidence_chunks(selected_pages)
|
| 997 |
+
chunk_by_id = {chunk.chunk_id: chunk for chunk in chunks}
|
| 998 |
+
claims: list[GroundedClaim] = []
|
| 999 |
+
for index, item in enumerate(claims_payload):
|
| 1000 |
+
if not isinstance(item, dict) or set(item) != {"text", "chunk_id", "quote"}:
|
| 1001 |
+
return _validation_failure(
|
| 1002 |
+
raw,
|
| 1003 |
+
f"claim {index + 1} must contain exactly text, chunk_id, and quote",
|
| 1004 |
+
)
|
| 1005 |
+
if not all(isinstance(item[key], str) for key in item):
|
| 1006 |
+
return _validation_failure(raw, f"claim {index + 1} fields must be strings")
|
| 1007 |
+
claim = GroundedClaim(
|
| 1008 |
+
text=item["text"].strip(),
|
| 1009 |
+
chunk_id=item["chunk_id"].strip(),
|
| 1010 |
+
quote=item["quote"].strip(),
|
| 1011 |
+
)
|
| 1012 |
+
reference_error = _validate_claim_reference(claim, chunk_by_id)
|
| 1013 |
+
if reference_error:
|
| 1014 |
+
return _validation_failure(raw, f"claim {index + 1}: {reference_error}")
|
| 1015 |
+
text_error = _validate_claim_text(claim)
|
| 1016 |
+
if text_error:
|
| 1017 |
+
return _validation_failure(raw, f"claim {index + 1}: {text_error}")
|
| 1018 |
+
safe_text = _normalize_text(claim.quote)
|
| 1019 |
+
if not re.search(r"[.!?][\"')\]]*$", safe_text):
|
| 1020 |
+
safe_text = f"{safe_text}."
|
| 1021 |
+
claim = replace(claim, text=safe_text)
|
| 1022 |
+
claims.append(claim)
|
| 1023 |
+
|
| 1024 |
+
query_error = _validate_query_grounding(
|
| 1025 |
+
query,
|
| 1026 |
+
claims,
|
| 1027 |
+
chunk_by_id,
|
| 1028 |
+
target_offer_ids or frozenset(),
|
| 1029 |
+
)
|
| 1030 |
+
if query_error:
|
| 1031 |
+
return _validation_failure(raw, query_error)
|
| 1032 |
+
|
| 1033 |
+
cited_chunks: list[EvidenceChunk] = []
|
| 1034 |
+
cited_ids: set[str] = set()
|
| 1035 |
+
for claim in claims:
|
| 1036 |
+
if claim.chunk_id not in cited_ids:
|
| 1037 |
+
cited_chunks.append(chunk_by_id[claim.chunk_id])
|
| 1038 |
+
cited_ids.add(claim.chunk_id)
|
| 1039 |
+
|
| 1040 |
+
return GroundingResult(
|
| 1041 |
+
valid=True,
|
| 1042 |
+
status="answered",
|
| 1043 |
+
answer=" ".join(claim.text for claim in claims),
|
| 1044 |
+
claims=tuple(claims),
|
| 1045 |
+
cited_chunks=tuple(cited_chunks),
|
| 1046 |
+
usage=raw.usage,
|
| 1047 |
+
latency_ms=raw.latency_ms,
|
| 1048 |
+
)
|
| 1049 |
+
|
| 1050 |
+
|
| 1051 |
+
def extract_mentorship_course_access(
|
| 1052 |
+
query: str,
|
| 1053 |
+
selected_pages: list[dict[str, Any]],
|
| 1054 |
+
*,
|
| 1055 |
+
target_offer_ids: frozenset[str] | None = None,
|
| 1056 |
+
) -> GroundingResult | None:
|
| 1057 |
+
"""Extract high-risk mentorship entitlements and policies without generation."""
|
| 1058 |
+
|
| 1059 |
+
lowered = query.casefold()
|
| 1060 |
+
if "mentor" not in lowered:
|
| 1061 |
+
return None
|
| 1062 |
+
|
| 1063 |
+
chunks = evidence_chunks(selected_pages)
|
| 1064 |
+
|
| 1065 |
+
def find_span(*phrases: str) -> tuple[EvidenceChunk, EvidenceSpan] | None:
|
| 1066 |
+
for chunk in chunks:
|
| 1067 |
+
if chunk.offer_id != "mentorship":
|
| 1068 |
+
continue
|
| 1069 |
+
for span in chunk.evidence_spans:
|
| 1070 |
+
span_lower = span.text.casefold()
|
| 1071 |
+
if all(phrase in span_lower for phrase in phrases):
|
| 1072 |
+
return chunk, span
|
| 1073 |
+
return None
|
| 1074 |
+
|
| 1075 |
+
def validated_extraction(
|
| 1076 |
+
evidence: list[tuple[EvidenceChunk, EvidenceSpan]],
|
| 1077 |
+
) -> GroundingResult:
|
| 1078 |
+
if not evidence:
|
| 1079 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1080 |
+
raw = json.dumps(
|
| 1081 |
+
{
|
| 1082 |
+
"status": "answered",
|
| 1083 |
+
"claims": [
|
| 1084 |
+
{
|
| 1085 |
+
"text": span.text,
|
| 1086 |
+
"chunk_id": chunk.chunk_id,
|
| 1087 |
+
"quote": span.text,
|
| 1088 |
+
}
|
| 1089 |
+
for chunk, span in evidence
|
| 1090 |
+
],
|
| 1091 |
+
},
|
| 1092 |
+
ensure_ascii=False,
|
| 1093 |
+
)
|
| 1094 |
+
result = validate_grounded_result(
|
| 1095 |
+
LLMResult(raw, usage={"provider": "deterministic_extraction"}),
|
| 1096 |
+
selected_pages,
|
| 1097 |
+
target_offer_ids=target_offer_ids,
|
| 1098 |
+
)
|
| 1099 |
+
return (
|
| 1100 |
+
result
|
| 1101 |
+
if result.valid
|
| 1102 |
+
else GroundingResult(valid=True, status="not_found")
|
| 1103 |
+
)
|
| 1104 |
+
|
| 1105 |
+
course_access_intent = any(
|
| 1106 |
+
term in lowered
|
| 1107 |
+
for term in (
|
| 1108 |
+
"course",
|
| 1109 |
+
"courses",
|
| 1110 |
+
"llm fundamentals",
|
| 1111 |
+
"full stack",
|
| 1112 |
+
"agent engineering",
|
| 1113 |
+
"master ai for work",
|
| 1114 |
+
)
|
| 1115 |
+
) and any(
|
| 1116 |
+
term in lowered
|
| 1117 |
+
for term in (
|
| 1118 |
+
"include",
|
| 1119 |
+
"included",
|
| 1120 |
+
"access",
|
| 1121 |
+
"choice",
|
| 1122 |
+
"choose",
|
| 1123 |
+
"discount",
|
| 1124 |
+
"% off",
|
| 1125 |
+
"free",
|
| 1126 |
+
"2",
|
| 1127 |
+
"two",
|
| 1128 |
+
)
|
| 1129 |
+
)
|
| 1130 |
+
cancellation_intent = bool(
|
| 1131 |
+
re.search(r"\bcancel|\bcancell|\bafter\b", lowered)
|
| 1132 |
+
and any(term in lowered for term in ("course", "access", "lifetime", "keep"))
|
| 1133 |
+
)
|
| 1134 |
+
if cancellation_intent:
|
| 1135 |
+
active = find_span("course access is active while")
|
| 1136 |
+
upgrade = find_span("if you cancel", "lifetime access")
|
| 1137 |
+
if active is None or upgrade is None:
|
| 1138 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1139 |
+
return validated_extraction([active, upgrade])
|
| 1140 |
+
|
| 1141 |
+
guarantee_intent = any(
|
| 1142 |
+
term in lowered for term in ("refund", "money-back", "money back", "guarantee")
|
| 1143 |
+
)
|
| 1144 |
+
if guarantee_intent:
|
| 1145 |
+
yearly = find_span("yearly plan", "money-back guarantee")
|
| 1146 |
+
if yearly is None:
|
| 1147 |
+
yearly = find_span("money-back on yearly")
|
| 1148 |
+
if monthly_plan_intent(lowered):
|
| 1149 |
+
monthly = find_span("monthly mentorship", "cancelled")
|
| 1150 |
+
if monthly is None or yearly is None:
|
| 1151 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1152 |
+
return validated_extraction([monthly, yearly])
|
| 1153 |
+
if yearly is None:
|
| 1154 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1155 |
+
return validated_extraction([yearly])
|
| 1156 |
+
|
| 1157 |
+
if not course_access_intent:
|
| 1158 |
+
return None
|
| 1159 |
+
|
| 1160 |
+
included: tuple[EvidenceChunk, EvidenceSpan] | None = None
|
| 1161 |
+
discounted: tuple[EvidenceChunk, EvidenceSpan] | None = None
|
| 1162 |
+
alpha: tuple[EvidenceChunk, EvidenceSpan] | None = None
|
| 1163 |
+
for chunk in chunks:
|
| 1164 |
+
if "/academy/mentorship" not in chunk.url:
|
| 1165 |
+
continue
|
| 1166 |
+
for span in chunk.evidence_spans:
|
| 1167 |
+
span_lower = span.text.casefold()
|
| 1168 |
+
if (
|
| 1169 |
+
included is None
|
| 1170 |
+
and "llm fundamentals" in span_lower
|
| 1171 |
+
and re.search(r"\bincluded\b", span_lower)
|
| 1172 |
+
):
|
| 1173 |
+
included = (chunk, span)
|
| 1174 |
+
if (
|
| 1175 |
+
discounted is None
|
| 1176 |
+
and any(
|
| 1177 |
+
name in span_lower
|
| 1178 |
+
for name in (
|
| 1179 |
+
"full stack ai engineering",
|
| 1180 |
+
"agent engineering",
|
| 1181 |
+
"master ai for work",
|
| 1182 |
+
)
|
| 1183 |
+
)
|
| 1184 |
+
and re.search(r"\b\d+(?:[–-]\d+)?%\s+off\b", span_lower)
|
| 1185 |
+
):
|
| 1186 |
+
discounted = (chunk, span)
|
| 1187 |
+
if (
|
| 1188 |
+
alpha is None
|
| 1189 |
+
and "new courses" in span_lower
|
| 1190 |
+
and "alpha access" in span_lower
|
| 1191 |
+
and "% off" in span_lower
|
| 1192 |
+
):
|
| 1193 |
+
alpha = (chunk, span)
|
| 1194 |
+
|
| 1195 |
+
if any(term in lowered for term in ("new course", "new courses", "alpha")):
|
| 1196 |
+
return (
|
| 1197 |
+
validated_extraction([alpha])
|
| 1198 |
+
if alpha is not None
|
| 1199 |
+
else GroundingResult(valid=True, status="not_found")
|
| 1200 |
+
)
|
| 1201 |
+
|
| 1202 |
+
wants_discount = any(
|
| 1203 |
+
term in lowered for term in ("discount", "% off", "off full", "off agent")
|
| 1204 |
+
) or bool(re.search(r"\b\d+(?:[.-]\d+)?%", lowered))
|
| 1205 |
+
wants_comparison = any(
|
| 1206 |
+
term in lowered
|
| 1207 |
+
for term in (
|
| 1208 |
+
"courses",
|
| 1209 |
+
"choice",
|
| 1210 |
+
"choose",
|
| 1211 |
+
"2",
|
| 1212 |
+
"two",
|
| 1213 |
+
"full stack",
|
| 1214 |
+
"agent engineering",
|
| 1215 |
+
"master ai for work",
|
| 1216 |
+
)
|
| 1217 |
+
)
|
| 1218 |
+
if wants_discount and discounted is None:
|
| 1219 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1220 |
+
if not wants_discount and included is None:
|
| 1221 |
+
return GroundingResult(valid=True, status="not_found")
|
| 1222 |
+
|
| 1223 |
+
evidence: list[tuple[EvidenceChunk, EvidenceSpan]] = []
|
| 1224 |
+
if included is not None and (not wants_discount or wants_comparison):
|
| 1225 |
+
evidence.append(included)
|
| 1226 |
+
if discounted is not None and (wants_discount or wants_comparison):
|
| 1227 |
+
evidence.append(discounted)
|
| 1228 |
+
return validated_extraction(evidence)
|
| 1229 |
+
|
| 1230 |
+
|
| 1231 |
+
def generate_grounded_answer(
|
| 1232 |
+
prompt: str,
|
| 1233 |
+
selected_pages: list[dict[str, Any]],
|
| 1234 |
+
*,
|
| 1235 |
+
query: str = "",
|
| 1236 |
+
target_offer_ids: frozenset[str] | None = None,
|
| 1237 |
+
) -> GroundingResult:
|
| 1238 |
+
"""Generate and cross the fail-closed grounding boundary in one call."""
|
| 1239 |
+
|
| 1240 |
+
return validate_grounded_result(
|
| 1241 |
+
generate_answer(prompt),
|
| 1242 |
+
selected_pages,
|
| 1243 |
+
query=query,
|
| 1244 |
+
target_offer_ids=target_offer_ids,
|
| 1245 |
)
|
tai_helper/monitoring.py
CHANGED
|
@@ -25,9 +25,11 @@ def _import_opik() -> Any | None:
|
|
| 25 |
global _OPIK_IMPORT_WARNING_EMITTED
|
| 26 |
try:
|
| 27 |
import opik # type: ignore[import-not-found]
|
| 28 |
-
except
|
| 29 |
if not _OPIK_IMPORT_WARNING_EMITTED:
|
| 30 |
-
logger.warning(
|
|
|
|
|
|
|
| 31 |
_OPIK_IMPORT_WARNING_EMITTED = True
|
| 32 |
return None
|
| 33 |
return opik
|
|
@@ -120,8 +122,8 @@ class HelperMonitor:
|
|
| 120 |
},
|
| 121 |
"tags": tags,
|
| 122 |
"project_name": self.configured.opik_project_name,
|
| 123 |
-
"model": self.configured.model_name,
|
| 124 |
-
"provider": "
|
| 125 |
"flush": True,
|
| 126 |
}
|
| 127 |
with opik.start_as_current_span(**span_kwargs) as span:
|
|
|
|
| 25 |
global _OPIK_IMPORT_WARNING_EMITTED
|
| 26 |
try:
|
| 27 |
import opik # type: ignore[import-not-found]
|
| 28 |
+
except ImportError:
|
| 29 |
if not _OPIK_IMPORT_WARNING_EMITTED:
|
| 30 |
+
logger.warning(
|
| 31 |
+
"Opik monitoring is enabled but the opik package is unavailable."
|
| 32 |
+
)
|
| 33 |
_OPIK_IMPORT_WARNING_EMITTED = True
|
| 34 |
return None
|
| 35 |
return opik
|
|
|
|
| 122 |
},
|
| 123 |
"tags": tags,
|
| 124 |
"project_name": self.configured.opik_project_name,
|
| 125 |
+
"model": self.usage.get("model") or self.configured.model_name,
|
| 126 |
+
"provider": self.usage.get("provider") or "unknown",
|
| 127 |
"flush": True,
|
| 128 |
}
|
| 129 |
with opik.start_as_current_span(**span_kwargs) as span:
|
tai_helper/schemas.py
CHANGED
|
@@ -36,4 +36,7 @@ class HelperChatResponse(BaseModel):
|
|
| 36 |
answer: str
|
| 37 |
threadId: str
|
| 38 |
sources: list[SourceOut]
|
|
|
|
|
|
|
|
|
|
| 39 |
usage: dict[str, Any] = Field(default_factory=dict)
|
|
|
|
| 36 |
answer: str
|
| 37 |
threadId: str
|
| 38 |
sources: list[SourceOut]
|
| 39 |
+
status: Literal["answered", "insufficient_evidence", "out_of_scope", "policy"] = (
|
| 40 |
+
"answered"
|
| 41 |
+
)
|
| 42 |
usage: dict[str, Any] = Field(default_factory=dict)
|
tai_helper/settings.py
CHANGED
|
@@ -28,6 +28,16 @@ def _int_env(name: str, default: int) -> int:
|
|
| 28 |
return default
|
| 29 |
|
| 30 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 31 |
def _bool_env(name: str, default: bool = False) -> bool:
|
| 32 |
raw = os.getenv(name)
|
| 33 |
if raw is None:
|
|
@@ -40,25 +50,78 @@ class Settings:
|
|
| 40 |
allowed_origins: tuple[str, ...] = field(
|
| 41 |
default_factory=lambda: _csv_env(
|
| 42 |
"HELPER_ALLOWED_ORIGINS",
|
| 43 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
)
|
| 45 |
)
|
| 46 |
allowed_hosts: tuple[str, ...] = field(
|
| 47 |
default_factory=lambda: _csv_env(
|
| 48 |
"HELPER_ALLOWED_HOSTS",
|
| 49 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 50 |
)
|
| 51 |
)
|
| 52 |
-
|
| 53 |
-
default_factory=lambda: os.getenv("
|
| 54 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 55 |
)
|
| 56 |
gemini_api_key: str = field(
|
| 57 |
default_factory=lambda: os.getenv("GEMINI_API_KEY", "").strip()
|
| 58 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 59 |
max_output_tokens: int = field(
|
| 60 |
default_factory=lambda: _int_env("HELPER_MAX_OUTPUT_TOKENS", 420)
|
| 61 |
)
|
|
|
|
|
|
|
|
|
|
| 62 |
max_body_bytes: int = field(
|
| 63 |
default_factory=lambda: _int_env("HELPER_MAX_BODY_BYTES", 64 * 1024)
|
| 64 |
)
|
|
@@ -68,6 +131,9 @@ class Settings:
|
|
| 68 |
max_history_turns: int = field(
|
| 69 |
default_factory=lambda: _int_env("HELPER_MAX_HISTORY_TURNS", 8)
|
| 70 |
)
|
|
|
|
|
|
|
|
|
|
| 71 |
rate_limit_per_minute: int = field(
|
| 72 |
default_factory=lambda: _int_env("HELPER_RATE_LIMIT_PER_MINUTE", 3)
|
| 73 |
)
|
|
@@ -85,11 +151,13 @@ class Settings:
|
|
| 85 |
default_factory=lambda: os.getenv("OPIK_WORKSPACE", "").strip()
|
| 86 |
)
|
| 87 |
opik_project_name: str = field(
|
| 88 |
-
default_factory=lambda:
|
| 89 |
-
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
|
|
|
|
|
|
| 93 |
)
|
| 94 |
opik_max_text_chars: int = field(
|
| 95 |
default_factory=lambda: _int_env("OPIK_MAX_TEXT_CHARS", 4000)
|
|
@@ -98,5 +166,9 @@ class Settings:
|
|
| 98 |
def cors_origins(self) -> list[str]:
|
| 99 |
return list(self.allowed_origins)
|
| 100 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 101 |
|
| 102 |
settings = Settings()
|
|
|
|
| 28 |
return default
|
| 29 |
|
| 30 |
|
| 31 |
+
def _float_env(name: str, default: float) -> float:
|
| 32 |
+
raw = os.getenv(name)
|
| 33 |
+
if raw is None:
|
| 34 |
+
return default
|
| 35 |
+
try:
|
| 36 |
+
return float(raw)
|
| 37 |
+
except ValueError:
|
| 38 |
+
return default
|
| 39 |
+
|
| 40 |
+
|
| 41 |
def _bool_env(name: str, default: bool = False) -> bool:
|
| 42 |
raw = os.getenv(name)
|
| 43 |
if raw is None:
|
|
|
|
| 50 |
allowed_origins: tuple[str, ...] = field(
|
| 51 |
default_factory=lambda: _csv_env(
|
| 52 |
"HELPER_ALLOWED_ORIGINS",
|
| 53 |
+
(
|
| 54 |
+
"https://towardsai.com,https://www.towardsai.com,"
|
| 55 |
+
"https://academy.towardsai.net,https://towardsai.net,"
|
| 56 |
+
"https://www.towardsai.net"
|
| 57 |
+
),
|
| 58 |
)
|
| 59 |
)
|
| 60 |
allowed_hosts: tuple[str, ...] = field(
|
| 61 |
default_factory=lambda: _csv_env(
|
| 62 |
"HELPER_ALLOWED_HOSTS",
|
| 63 |
+
(
|
| 64 |
+
"towardsai.com,www.towardsai.com,academy.towardsai.net,"
|
| 65 |
+
"towardsai.net,www.towardsai.net"
|
| 66 |
+
),
|
| 67 |
+
)
|
| 68 |
+
)
|
| 69 |
+
site_wide_hosts: tuple[str, ...] = field(
|
| 70 |
+
default_factory=lambda: _csv_env(
|
| 71 |
+
"HELPER_SITE_WIDE_HOSTS",
|
| 72 |
+
"towardsai.com,www.towardsai.com",
|
| 73 |
)
|
| 74 |
)
|
| 75 |
+
deepseek_api_key: str = field(
|
| 76 |
+
default_factory=lambda: os.getenv("DEEPSEEK_API_KEY", "").strip()
|
| 77 |
+
)
|
| 78 |
+
deepseek_base_url: str = field(
|
| 79 |
+
default_factory=lambda: (
|
| 80 |
+
os.getenv(
|
| 81 |
+
"DEEPSEEK_BASE_URL",
|
| 82 |
+
"https://api.deepseek.com",
|
| 83 |
+
)
|
| 84 |
+
.strip()
|
| 85 |
+
.rstrip("/")
|
| 86 |
+
or "https://api.deepseek.com"
|
| 87 |
+
)
|
| 88 |
+
)
|
| 89 |
+
deepseek_thinking_type: str = field(
|
| 90 |
+
default_factory=lambda: (
|
| 91 |
+
os.getenv(
|
| 92 |
+
"HELPER_DEEPSEEK_THINKING",
|
| 93 |
+
"disabled",
|
| 94 |
+
).strip()
|
| 95 |
+
or "disabled"
|
| 96 |
+
)
|
| 97 |
+
)
|
| 98 |
+
primary_model_name: str = field(
|
| 99 |
+
default_factory=lambda: (
|
| 100 |
+
os.getenv(
|
| 101 |
+
"HELPER_PRIMARY_MODEL",
|
| 102 |
+
os.getenv("HELPER_MODEL", "deepseek-v4-flash"),
|
| 103 |
+
).strip()
|
| 104 |
+
or "deepseek-v4-flash"
|
| 105 |
+
)
|
| 106 |
)
|
| 107 |
gemini_api_key: str = field(
|
| 108 |
default_factory=lambda: os.getenv("GEMINI_API_KEY", "").strip()
|
| 109 |
)
|
| 110 |
+
fallback_model_name: str = field(
|
| 111 |
+
default_factory=lambda: (
|
| 112 |
+
os.getenv(
|
| 113 |
+
"HELPER_FALLBACK_MODEL",
|
| 114 |
+
"gemini-2.5-flash",
|
| 115 |
+
).strip()
|
| 116 |
+
or "gemini-2.5-flash"
|
| 117 |
+
)
|
| 118 |
+
)
|
| 119 |
max_output_tokens: int = field(
|
| 120 |
default_factory=lambda: _int_env("HELPER_MAX_OUTPUT_TOKENS", 420)
|
| 121 |
)
|
| 122 |
+
llm_request_timeout_seconds: float = field(
|
| 123 |
+
default_factory=lambda: _float_env("HELPER_LLM_REQUEST_TIMEOUT_SECONDS", 20.0)
|
| 124 |
+
)
|
| 125 |
max_body_bytes: int = field(
|
| 126 |
default_factory=lambda: _int_env("HELPER_MAX_BODY_BYTES", 64 * 1024)
|
| 127 |
)
|
|
|
|
| 131 |
max_history_turns: int = field(
|
| 132 |
default_factory=lambda: _int_env("HELPER_MAX_HISTORY_TURNS", 8)
|
| 133 |
)
|
| 134 |
+
catalog_max_age_days: int = field(
|
| 135 |
+
default_factory=lambda: _int_env("HELPER_CATALOG_MAX_AGE_DAYS", 14)
|
| 136 |
+
)
|
| 137 |
rate_limit_per_minute: int = field(
|
| 138 |
default_factory=lambda: _int_env("HELPER_RATE_LIMIT_PER_MINUTE", 3)
|
| 139 |
)
|
|
|
|
| 151 |
default_factory=lambda: os.getenv("OPIK_WORKSPACE", "").strip()
|
| 152 |
)
|
| 153 |
opik_project_name: str = field(
|
| 154 |
+
default_factory=lambda: (
|
| 155 |
+
os.getenv(
|
| 156 |
+
"OPIK_PROJECT_NAME",
|
| 157 |
+
"towards-ai-helper",
|
| 158 |
+
).strip()
|
| 159 |
+
or "towards-ai-helper"
|
| 160 |
+
)
|
| 161 |
)
|
| 162 |
opik_max_text_chars: int = field(
|
| 163 |
default_factory=lambda: _int_env("OPIK_MAX_TEXT_CHARS", 4000)
|
|
|
|
| 166 |
def cors_origins(self) -> list[str]:
|
| 167 |
return list(self.allowed_origins)
|
| 168 |
|
| 169 |
+
@property
|
| 170 |
+
def model_name(self) -> str:
|
| 171 |
+
return self.primary_model_name
|
| 172 |
+
|
| 173 |
|
| 174 |
settings = Settings()
|
tests/conftest.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import os
|
| 4 |
+
|
| 5 |
+
os.environ["HELPER_ALLOWED_ORIGINS"] = (
|
| 6 |
+
"https://towardsai.com,https://www.towardsai.com,"
|
| 7 |
+
"https://academy.towardsai.net,https://towardsai.net,"
|
| 8 |
+
"https://www.towardsai.net"
|
| 9 |
+
)
|
| 10 |
+
os.environ["HELPER_ALLOWED_HOSTS"] = (
|
| 11 |
+
"towardsai.com,www.towardsai.com,academy.towardsai.net,"
|
| 12 |
+
"towardsai.net,www.towardsai.net"
|
| 13 |
+
)
|
| 14 |
+
os.environ["HELPER_SITE_WIDE_HOSTS"] = "towardsai.com,www.towardsai.com"
|
tests/test_api.py
CHANGED
|
@@ -1,14 +1,17 @@
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
|
|
|
|
|
|
|
|
|
| 3 |
from fastapi.testclient import TestClient
|
| 4 |
|
| 5 |
from tai_helper import api
|
| 6 |
-
from tai_helper.llm import
|
| 7 |
from tai_helper.rate_limiter import FixedWindowRateLimiter, RateLimit
|
| 8 |
|
| 9 |
client = TestClient(api.app)
|
| 10 |
-
HEADERS = {"Origin": "https://
|
| 11 |
-
PUBLIC_URL = "https://
|
| 12 |
FIRST_PROMPT = "I want help deciding which course to take."
|
| 13 |
|
| 14 |
|
|
@@ -30,7 +33,9 @@ def reset_limiters() -> None:
|
|
| 30 |
)
|
| 31 |
|
| 32 |
|
| 33 |
-
def payload(
|
|
|
|
|
|
|
| 34 |
return {
|
| 35 |
"query": query,
|
| 36 |
"selectedPrompt": query if query == FIRST_PROMPT else FIRST_PROMPT,
|
|
@@ -45,6 +50,13 @@ def payload(query: str = FIRST_PROMPT, *, url: str = PUBLIC_URL, signed_in: bool
|
|
| 45 |
}
|
| 46 |
|
| 47 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
def test_config_exposes_public_widget_contract() -> None:
|
| 49 |
response = client.get("/api/helper/config")
|
| 50 |
|
|
@@ -52,7 +64,33 @@ def test_config_exposes_public_widget_contract() -> None:
|
|
| 52 |
data = response.json()
|
| 53 |
assert data["name"] == "Towards AI Helper"
|
| 54 |
assert FIRST_PROMPT in data["forcedPrompts"]
|
| 55 |
-
assert "
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
|
| 57 |
|
| 58 |
def test_chat_requires_allowed_origin_public_page_signed_out_and_first_prompt() -> None:
|
|
@@ -62,7 +100,7 @@ def test_chat_requires_allowed_origin_public_page_signed_out_and_first_prompt()
|
|
| 62 |
assert (
|
| 63 |
client.post(
|
| 64 |
"/api/helper/chat",
|
| 65 |
-
json=payload(url="https://
|
| 66 |
headers=HEADERS,
|
| 67 |
).status_code
|
| 68 |
== 403
|
|
@@ -85,38 +123,506 @@ def test_chat_requires_allowed_origin_public_page_signed_out_and_first_prompt()
|
|
| 85 |
)
|
| 86 |
|
| 87 |
|
| 88 |
-
def
|
| 89 |
reset_limiters()
|
| 90 |
prompts = []
|
| 91 |
|
| 92 |
-
def
|
|
|
|
|
|
|
| 93 |
prompts.append(prompt)
|
| 94 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 95 |
answer="Tell me your coding background and goal.",
|
|
|
|
| 96 |
usage={"input_tokens": 10, "output_tokens": 8, "total_tokens": 18},
|
| 97 |
latency_ms=123,
|
| 98 |
)
|
| 99 |
|
| 100 |
-
monkeypatch.setattr(
|
|
|
|
|
|
|
| 101 |
|
| 102 |
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 103 |
|
| 104 |
assert response.status_code == 200
|
| 105 |
data = response.json()
|
| 106 |
assert data["answer"] == "Tell me your coding background and goal."
|
|
|
|
| 107 |
assert data["threadId"]
|
| 108 |
assert data["sources"]
|
| 109 |
assert data["usage"]["total_tokens"] == 18
|
| 110 |
assert "Agent Engineering" in prompts[0]
|
| 111 |
|
| 112 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 113 |
def test_coupon_answer_is_deterministic_and_does_not_call_model(monkeypatch) -> None:
|
| 114 |
reset_limiters()
|
| 115 |
|
| 116 |
-
def unexpected_generate_answer(
|
|
|
|
|
|
|
| 117 |
raise AssertionError("model should not be called for coupon intent")
|
| 118 |
|
| 119 |
-
monkeypatch.setattr(api.llm, "
|
| 120 |
first = payload(query="Do you have a coupon code?")
|
| 121 |
first["selectedPrompt"] = FIRST_PROMPT
|
| 122 |
first["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
|
@@ -131,9 +637,10 @@ def test_coupon_answer_is_deterministic_and_does_not_call_model(monkeypatch) ->
|
|
| 131 |
second_response = client.post("/api/helper/chat", json=second, headers=HEADERS)
|
| 132 |
|
| 133 |
assert first_response.status_code == 200
|
| 134 |
-
assert
|
|
|
|
| 135 |
assert second_response.status_code == 200
|
| 136 |
-
assert
|
| 137 |
|
| 138 |
|
| 139 |
def test_rate_limit_is_hard(monkeypatch) -> None:
|
|
@@ -145,8 +652,22 @@ def test_rate_limit_is_hard(monkeypatch) -> None:
|
|
| 145 |
|
| 146 |
monkeypatch.setattr(
|
| 147 |
api.llm,
|
| 148 |
-
"
|
| 149 |
-
lambda _prompt:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 150 |
)
|
| 151 |
|
| 152 |
first = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
|
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
| 3 |
+
import json
|
| 4 |
+
|
| 5 |
+
import pytest
|
| 6 |
from fastapi.testclient import TestClient
|
| 7 |
|
| 8 |
from tai_helper import api
|
| 9 |
+
from tai_helper.llm import EvidenceChunk, GroundingResult
|
| 10 |
from tai_helper.rate_limiter import FixedWindowRateLimiter, RateLimit
|
| 11 |
|
| 12 |
client = TestClient(api.app)
|
| 13 |
+
HEADERS = {"Origin": "https://towardsai.com"}
|
| 14 |
+
PUBLIC_URL = "https://towardsai.com/academy/agent-engineering/"
|
| 15 |
FIRST_PROMPT = "I want help deciding which course to take."
|
| 16 |
|
| 17 |
|
|
|
|
| 33 |
)
|
| 34 |
|
| 35 |
|
| 36 |
+
def payload(
|
| 37 |
+
query: str = FIRST_PROMPT, *, url: str = PUBLIC_URL, signed_in: bool = False
|
| 38 |
+
):
|
| 39 |
return {
|
| 40 |
"query": query,
|
| 41 |
"selectedPrompt": query if query == FIRST_PROMPT else FIRST_PROMPT,
|
|
|
|
| 50 |
}
|
| 51 |
|
| 52 |
|
| 53 |
+
def assert_contact_handoff(body: dict, *, status: str = "insufficient_evidence") -> None:
|
| 54 |
+
assert body["status"] == status
|
| 55 |
+
assert body["sources"] == []
|
| 56 |
+
assert body["answer"].count(api.CONTACT_FORM_URL) == 1
|
| 57 |
+
assert "louis@towardsai.net" not in body["answer"]
|
| 58 |
+
|
| 59 |
+
|
| 60 |
def test_config_exposes_public_widget_contract() -> None:
|
| 61 |
response = client.get("/api/helper/config")
|
| 62 |
|
|
|
|
| 64 |
data = response.json()
|
| 65 |
assert data["name"] == "Towards AI Helper"
|
| 66 |
assert FIRST_PROMPT in data["forcedPrompts"]
|
| 67 |
+
assert "towardsai.com" in data["allowedHosts"]
|
| 68 |
+
assert "towardsai.com" in data["siteWideHosts"]
|
| 69 |
+
assert "/academy/agent-engineering" in data["allowedPathsByHost"]["towardsai.com"]
|
| 70 |
+
assert (
|
| 71 |
+
"/courses/agent-engineering"
|
| 72 |
+
in data["allowedPathsByHost"]["academy.towardsai.net"]
|
| 73 |
+
)
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def test_footer_compatible_widget_path_is_served() -> None:
|
| 77 |
+
response = client.get("/helper-widget.js")
|
| 78 |
+
|
| 79 |
+
assert response.status_code == 200
|
| 80 |
+
assert "Towards AI Helper" in response.text
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
def test_cors_preflight_accepts_towardsai_com() -> None:
|
| 84 |
+
response = client.options(
|
| 85 |
+
"/api/helper/chat",
|
| 86 |
+
headers={
|
| 87 |
+
"Origin": "https://towardsai.com",
|
| 88 |
+
"Access-Control-Request-Method": "POST",
|
| 89 |
+
},
|
| 90 |
+
)
|
| 91 |
+
|
| 92 |
+
assert response.status_code == 200
|
| 93 |
+
assert response.headers["access-control-allow-origin"] == "https://towardsai.com"
|
| 94 |
|
| 95 |
|
| 96 |
def test_chat_requires_allowed_origin_public_page_signed_out_and_first_prompt() -> None:
|
|
|
|
| 100 |
assert (
|
| 101 |
client.post(
|
| 102 |
"/api/helper/chat",
|
| 103 |
+
json=payload(url="https://towardsai.com/wp-admin/edit.php"),
|
| 104 |
headers=HEADERS,
|
| 105 |
).status_code
|
| 106 |
== 403
|
|
|
|
| 123 |
)
|
| 124 |
|
| 125 |
|
| 126 |
+
def test_chat_returns_only_grounded_answer_and_cited_sources(monkeypatch) -> None:
|
| 127 |
reset_limiters()
|
| 128 |
prompts = []
|
| 129 |
|
| 130 |
+
def fake_generate_grounded_answer(
|
| 131 |
+
prompt: str, selected_pages: list[dict], **_kwargs
|
| 132 |
+
) -> GroundingResult:
|
| 133 |
prompts.append(prompt)
|
| 134 |
+
assert selected_pages
|
| 135 |
+
cited = EvidenceChunk(
|
| 136 |
+
chunk_id=str(selected_pages[0]["chunk_id"]),
|
| 137 |
+
title=str(selected_pages[0]["title"]),
|
| 138 |
+
url=str(selected_pages[0]["url"]),
|
| 139 |
+
kind=str(selected_pages[0]["kind"]),
|
| 140 |
+
headings=tuple(selected_pages[0].get("headings", [])),
|
| 141 |
+
text=str(selected_pages[0]["text"]),
|
| 142 |
+
)
|
| 143 |
+
return GroundingResult(
|
| 144 |
+
valid=True,
|
| 145 |
+
status="answered",
|
| 146 |
answer="Tell me your coding background and goal.",
|
| 147 |
+
cited_chunks=(cited,),
|
| 148 |
usage={"input_tokens": 10, "output_tokens": 8, "total_tokens": 18},
|
| 149 |
latency_ms=123,
|
| 150 |
)
|
| 151 |
|
| 152 |
+
monkeypatch.setattr(
|
| 153 |
+
api.llm, "generate_grounded_answer", fake_generate_grounded_answer
|
| 154 |
+
)
|
| 155 |
|
| 156 |
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 157 |
|
| 158 |
assert response.status_code == 200
|
| 159 |
data = response.json()
|
| 160 |
assert data["answer"] == "Tell me your coding background and goal."
|
| 161 |
+
assert data["status"] == "answered"
|
| 162 |
assert data["threadId"]
|
| 163 |
assert data["sources"]
|
| 164 |
assert data["usage"]["total_tokens"] == 18
|
| 165 |
assert "Agent Engineering" in prompts[0]
|
| 166 |
|
| 167 |
|
| 168 |
+
def test_chat_still_accepts_legacy_net_origin(monkeypatch) -> None:
|
| 169 |
+
reset_limiters()
|
| 170 |
+
monkeypatch.setattr(
|
| 171 |
+
api.llm,
|
| 172 |
+
"generate_grounded_answer",
|
| 173 |
+
lambda _prompt, selected, **_kwargs: GroundingResult(
|
| 174 |
+
valid=True,
|
| 175 |
+
status="answered",
|
| 176 |
+
answer="Supported answer.",
|
| 177 |
+
cited_chunks=(
|
| 178 |
+
EvidenceChunk(
|
| 179 |
+
chunk_id=str(selected[0]["chunk_id"]),
|
| 180 |
+
title=str(selected[0]["title"]),
|
| 181 |
+
url=str(selected[0]["url"]),
|
| 182 |
+
kind=str(selected[0]["kind"]),
|
| 183 |
+
headings=(),
|
| 184 |
+
text=str(selected[0]["text"]),
|
| 185 |
+
),
|
| 186 |
+
),
|
| 187 |
+
),
|
| 188 |
+
)
|
| 189 |
+
|
| 190 |
+
response = client.post(
|
| 191 |
+
"/api/helper/chat",
|
| 192 |
+
json=payload(url="https://towardsai.net/b2b"),
|
| 193 |
+
headers={"Origin": "https://towardsai.net"},
|
| 194 |
+
)
|
| 195 |
+
|
| 196 |
+
assert response.status_code == 200
|
| 197 |
+
|
| 198 |
+
|
| 199 |
+
def test_chat_abstains_without_retrieved_evidence(monkeypatch) -> None:
|
| 200 |
+
reset_limiters()
|
| 201 |
+
monkeypatch.setattr(api, "retrieve", lambda *_args, **_kwargs: [])
|
| 202 |
+
|
| 203 |
+
def unexpected_model(*_args, **_kwargs):
|
| 204 |
+
raise AssertionError("model must not run without retrieved evidence")
|
| 205 |
+
|
| 206 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected_model)
|
| 207 |
+
|
| 208 |
+
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 209 |
+
|
| 210 |
+
assert response.status_code == 200
|
| 211 |
+
assert_contact_handoff(response.json())
|
| 212 |
+
assert "don't want to guess" in response.json()["answer"]
|
| 213 |
+
|
| 214 |
+
|
| 215 |
+
def test_chat_abstains_on_model_not_found_or_validation_failure(monkeypatch) -> None:
|
| 216 |
+
reset_limiters()
|
| 217 |
+
evidence = {
|
| 218 |
+
"chunk_id": "mentorship-inclusions",
|
| 219 |
+
"title": "Towards AI Mentorship",
|
| 220 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 221 |
+
"kind": "mentorship",
|
| 222 |
+
"headings": ["Course included"],
|
| 223 |
+
"text": "10-Hour LLM Fundamentals video course — $199 — Included from day one.",
|
| 224 |
+
}
|
| 225 |
+
monkeypatch.setattr(api, "retrieve", lambda *_args, **_kwargs: [evidence])
|
| 226 |
+
|
| 227 |
+
for grounded in (
|
| 228 |
+
GroundingResult(valid=True, status="not_found"),
|
| 229 |
+
GroundingResult(
|
| 230 |
+
valid=False,
|
| 231 |
+
status="validation_failure",
|
| 232 |
+
validation_error="invented quote",
|
| 233 |
+
),
|
| 234 |
+
):
|
| 235 |
+
monkeypatch.setattr(
|
| 236 |
+
api.llm,
|
| 237 |
+
"generate_grounded_answer",
|
| 238 |
+
lambda *_args, result=grounded: result,
|
| 239 |
+
)
|
| 240 |
+
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 241 |
+
|
| 242 |
+
assert response.status_code == 200
|
| 243 |
+
assert_contact_handoff(response.json())
|
| 244 |
+
assert "don't want to guess" in response.json()["answer"]
|
| 245 |
+
|
| 246 |
+
|
| 247 |
+
def test_chat_abstains_when_model_omits_negation_from_an_evidence_span(
|
| 248 |
+
monkeypatch,
|
| 249 |
+
) -> None:
|
| 250 |
+
reset_limiters()
|
| 251 |
+
selected = api.retrieve(FIRST_PROMPT)
|
| 252 |
+
ai_work = next(page for page in selected if "No code required" in page["text"])
|
| 253 |
+
raw = json.dumps(
|
| 254 |
+
{
|
| 255 |
+
"status": "answered",
|
| 256 |
+
"claims": [
|
| 257 |
+
{
|
| 258 |
+
"text": "code required.",
|
| 259 |
+
"chunk_id": ai_work["chunk_id"],
|
| 260 |
+
"quote": "code required",
|
| 261 |
+
}
|
| 262 |
+
],
|
| 263 |
+
}
|
| 264 |
+
)
|
| 265 |
+
monkeypatch.setattr(
|
| 266 |
+
api.llm,
|
| 267 |
+
"generate_answer",
|
| 268 |
+
lambda _prompt: api.llm.LLMResult(answer=raw),
|
| 269 |
+
)
|
| 270 |
+
|
| 271 |
+
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 272 |
+
|
| 273 |
+
assert response.status_code == 200
|
| 274 |
+
assert_contact_handoff(response.json())
|
| 275 |
+
assert "don't want to guess" in response.json()["answer"]
|
| 276 |
+
|
| 277 |
+
|
| 278 |
+
@pytest.mark.parametrize(
|
| 279 |
+
"query",
|
| 280 |
+
[
|
| 281 |
+
(
|
| 282 |
+
"Do the 7 lessons in the Agent Engineering preview include "
|
| 283 |
+
"lifetime access?"
|
| 284 |
+
),
|
| 285 |
+
(
|
| 286 |
+
"Do the 7 lessons in the Agent Engineering preview have a "
|
| 287 |
+
"30-day refund guarantee?"
|
| 288 |
+
),
|
| 289 |
+
"Do the 7 lessons in the Agent Engineering preview give a certificate?",
|
| 290 |
+
"Do the 7 lessons in the Agent Engineering preview cost $499?",
|
| 291 |
+
"Do the 7 lessons in the Agent Engineering preview get a 50% discount?",
|
| 292 |
+
],
|
| 293 |
+
)
|
| 294 |
+
def test_preview_count_cannot_borrow_other_paid_course_facts(
|
| 295 |
+
monkeypatch, query: str
|
| 296 |
+
) -> None:
|
| 297 |
+
reset_limiters()
|
| 298 |
+
assert api.retrieve(query) == []
|
| 299 |
+
|
| 300 |
+
def unexpected(*_args, **_kwargs):
|
| 301 |
+
raise AssertionError("mixed preview facts must not invoke the model")
|
| 302 |
+
|
| 303 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected)
|
| 304 |
+
request_payload = payload(query=query)
|
| 305 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 306 |
+
|
| 307 |
+
response = client.post(
|
| 308 |
+
"/api/helper/chat", json=request_payload, headers=HEADERS
|
| 309 |
+
)
|
| 310 |
+
|
| 311 |
+
assert response.status_code == 200
|
| 312 |
+
assert_contact_handoff(response.json())
|
| 313 |
+
|
| 314 |
+
|
| 315 |
+
def test_preview_count_cannot_mask_an_unsupported_feature(monkeypatch) -> None:
|
| 316 |
+
reset_limiters()
|
| 317 |
+
query = (
|
| 318 |
+
"Do the 7 lessons in the Agent Engineering preview include weekly "
|
| 319 |
+
"code reviews?"
|
| 320 |
+
)
|
| 321 |
+
selected = api.retrieve(query)
|
| 322 |
+
count_chunk = next(
|
| 323 |
+
chunk
|
| 324 |
+
for chunk in selected
|
| 325 |
+
if any("7" in span["text"] for span in chunk["evidence_spans"])
|
| 326 |
+
)
|
| 327 |
+
count_span = next(
|
| 328 |
+
span["text"]
|
| 329 |
+
for span in count_chunk["evidence_spans"]
|
| 330 |
+
if "7" in span["text"]
|
| 331 |
+
)
|
| 332 |
+
raw = json.dumps(
|
| 333 |
+
{
|
| 334 |
+
"status": "answered",
|
| 335 |
+
"claims": [
|
| 336 |
+
{
|
| 337 |
+
"text": count_span,
|
| 338 |
+
"chunk_id": count_chunk["chunk_id"],
|
| 339 |
+
"quote": count_span,
|
| 340 |
+
}
|
| 341 |
+
],
|
| 342 |
+
}
|
| 343 |
+
)
|
| 344 |
+
monkeypatch.setattr(
|
| 345 |
+
api.llm,
|
| 346 |
+
"generate_answer",
|
| 347 |
+
lambda _prompt: api.llm.LLMResult(answer=raw),
|
| 348 |
+
)
|
| 349 |
+
request_payload = payload(query=query)
|
| 350 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 351 |
+
|
| 352 |
+
response = client.post(
|
| 353 |
+
"/api/helper/chat", json=request_payload, headers=HEADERS
|
| 354 |
+
)
|
| 355 |
+
|
| 356 |
+
assert response.status_code == 200
|
| 357 |
+
assert_contact_handoff(response.json())
|
| 358 |
+
|
| 359 |
+
|
| 360 |
+
def test_mentorship_course_access_incident_uses_exact_retrieved_rows(
|
| 361 |
+
monkeypatch,
|
| 362 |
+
) -> None:
|
| 363 |
+
reset_limiters()
|
| 364 |
+
|
| 365 |
+
def unexpected_model(*_args, **_kwargs):
|
| 366 |
+
raise AssertionError("high-risk mentorship access answer must be extractive")
|
| 367 |
+
|
| 368 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected_model)
|
| 369 |
+
request_payload = payload(
|
| 370 |
+
query=(
|
| 371 |
+
"The chatbot mentioned access to 2 courses of our choice as part of "
|
| 372 |
+
"mentorship. Is that true?"
|
| 373 |
+
),
|
| 374 |
+
url="https://towardsai.com/academy/mentorship/",
|
| 375 |
+
)
|
| 376 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 377 |
+
|
| 378 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 379 |
+
|
| 380 |
+
assert response.status_code == 200
|
| 381 |
+
body = response.json()
|
| 382 |
+
assert body["status"] == "answered"
|
| 383 |
+
assert "10-Hour LLM Fundamentals" in body["answer"]
|
| 384 |
+
assert "Included from day one" in body["answer"]
|
| 385 |
+
assert "25% off, always" in body["answer"]
|
| 386 |
+
assert "two courses" not in body["answer"].lower()
|
| 387 |
+
assert len(body["sources"]) == 1
|
| 388 |
+
assert body["sources"][0]["url"] == ("https://towardsai.com/academy/mentorship/")
|
| 389 |
+
assert body["sources"][0]["kind"] == "mentorship"
|
| 390 |
+
|
| 391 |
+
|
| 392 |
+
def test_mentorship_cancellation_uses_exact_current_access_policy(
|
| 393 |
+
monkeypatch,
|
| 394 |
+
) -> None:
|
| 395 |
+
reset_limiters()
|
| 396 |
+
|
| 397 |
+
def unexpected_model(*_args, **_kwargs):
|
| 398 |
+
raise AssertionError("mentorship cancellation answer must be extractive")
|
| 399 |
+
|
| 400 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected_model)
|
| 401 |
+
request_payload = payload(
|
| 402 |
+
query=(
|
| 403 |
+
"Do I keep LLM Fundamentals lifetime access after cancelling mentorship?"
|
| 404 |
+
),
|
| 405 |
+
url="https://towardsai.com/academy/mentorship/",
|
| 406 |
+
)
|
| 407 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 408 |
+
|
| 409 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 410 |
+
|
| 411 |
+
assert response.status_code == 200
|
| 412 |
+
body = response.json()
|
| 413 |
+
assert body["status"] == "answered"
|
| 414 |
+
assert "Course access is active while your mentorship is active" in body["answer"]
|
| 415 |
+
assert "If you cancel" in body["answer"]
|
| 416 |
+
assert "lifetime access at a reduced price" in body["answer"]
|
| 417 |
+
assert "Included from day one" not in body["answer"]
|
| 418 |
+
assert {source["url"] for source in body["sources"]} == {
|
| 419 |
+
"https://towardsai.com/academy/mentorship/"
|
| 420 |
+
}
|
| 421 |
+
|
| 422 |
+
|
| 423 |
+
@pytest.mark.parametrize(
|
| 424 |
+
"query",
|
| 425 |
+
[
|
| 426 |
+
"Does the monthly mentorship have a 30-day money-back guarantee?",
|
| 427 |
+
"Does the month-to-month mentorship have a 30-day money-back guarantee?",
|
| 428 |
+
],
|
| 429 |
+
)
|
| 430 |
+
def test_monthly_mentorship_guarantee_names_the_yearly_qualification(
|
| 431 |
+
monkeypatch, query: str
|
| 432 |
+
) -> None:
|
| 433 |
+
reset_limiters()
|
| 434 |
+
|
| 435 |
+
def unexpected_model(*_args, **_kwargs):
|
| 436 |
+
raise AssertionError("mentorship guarantee answer must be extractive")
|
| 437 |
+
|
| 438 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected_model)
|
| 439 |
+
request_payload = payload(
|
| 440 |
+
query=query,
|
| 441 |
+
url="https://towardsai.com/academy/mentorship/",
|
| 442 |
+
)
|
| 443 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 444 |
+
|
| 445 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 446 |
+
|
| 447 |
+
assert response.status_code == 200
|
| 448 |
+
body = response.json()
|
| 449 |
+
assert body["status"] == "answered"
|
| 450 |
+
assert "monthly mentorship can be cancelled at any time" in body["answer"]
|
| 451 |
+
assert "yearly plan carries a 30-day money-back guarantee" in body["answer"]
|
| 452 |
+
assert "Join the Mentorship 30-day" not in body["answer"]
|
| 453 |
+
assert len(body["sources"]) == 1
|
| 454 |
+
|
| 455 |
+
|
| 456 |
+
def test_late_source_error_resets_answered_status_to_insufficient_evidence(
|
| 457 |
+
monkeypatch,
|
| 458 |
+
) -> None:
|
| 459 |
+
reset_limiters()
|
| 460 |
+
selected = api.retrieve(FIRST_PROMPT)
|
| 461 |
+
cited = EvidenceChunk(
|
| 462 |
+
chunk_id=str(selected[0]["chunk_id"]),
|
| 463 |
+
title=str(selected[0]["title"]),
|
| 464 |
+
url=str(selected[0]["url"]),
|
| 465 |
+
kind=str(selected[0]["kind"]),
|
| 466 |
+
headings=tuple(selected[0].get("headings", [])),
|
| 467 |
+
text=str(selected[0]["text"]),
|
| 468 |
+
)
|
| 469 |
+
monkeypatch.setattr(
|
| 470 |
+
api.llm,
|
| 471 |
+
"generate_grounded_answer",
|
| 472 |
+
lambda *_args: GroundingResult(
|
| 473 |
+
valid=True,
|
| 474 |
+
status="answered",
|
| 475 |
+
answer="Verified source span.",
|
| 476 |
+
cited_chunks=(cited,),
|
| 477 |
+
),
|
| 478 |
+
)
|
| 479 |
+
monkeypatch.setattr(
|
| 480 |
+
api,
|
| 481 |
+
"sources_from_pages",
|
| 482 |
+
lambda _pages: (_ for _ in ()).throw(RuntimeError("late source failure")),
|
| 483 |
+
)
|
| 484 |
+
|
| 485 |
+
response = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
| 486 |
+
|
| 487 |
+
assert response.status_code == 200
|
| 488 |
+
assert_contact_handoff(response.json())
|
| 489 |
+
assert "don't want to guess" in response.json()["answer"]
|
| 490 |
+
|
| 491 |
+
|
| 492 |
+
def test_prior_assistant_claim_is_not_added_to_retrieval_query(monkeypatch) -> None:
|
| 493 |
+
reset_limiters()
|
| 494 |
+
captured = []
|
| 495 |
+
|
| 496 |
+
def fake_retrieve(query: str, **_kwargs):
|
| 497 |
+
captured.append(query)
|
| 498 |
+
return []
|
| 499 |
+
|
| 500 |
+
monkeypatch.setattr(api, "retrieve", fake_retrieve)
|
| 501 |
+
request_payload = payload(query="Is that true?")
|
| 502 |
+
request_payload["history"] = [
|
| 503 |
+
{"role": "user", "content": FIRST_PROMPT},
|
| 504 |
+
{
|
| 505 |
+
"role": "assistant",
|
| 506 |
+
"content": "The mentorship includes two courses of your choice.",
|
| 507 |
+
},
|
| 508 |
+
]
|
| 509 |
+
|
| 510 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 511 |
+
|
| 512 |
+
assert response.status_code == 200
|
| 513 |
+
assert captured
|
| 514 |
+
assert "two courses" not in captured[0].lower()
|
| 515 |
+
assert response.json()["status"] == "insufficient_evidence"
|
| 516 |
+
|
| 517 |
+
|
| 518 |
+
def test_substantive_followup_drops_starter_prompt_from_retrieval(monkeypatch) -> None:
|
| 519 |
+
reset_limiters()
|
| 520 |
+
captured: list[str] = []
|
| 521 |
+
query = "Does LLM Fundamentals include community support and an AI tutor?"
|
| 522 |
+
|
| 523 |
+
def fake_retrieve(retrieval_query: str, **_kwargs):
|
| 524 |
+
captured.append(retrieval_query)
|
| 525 |
+
return []
|
| 526 |
+
|
| 527 |
+
monkeypatch.setattr(api, "retrieve", fake_retrieve)
|
| 528 |
+
request_payload = payload(query=query)
|
| 529 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 530 |
+
|
| 531 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 532 |
+
|
| 533 |
+
assert response.status_code == 200
|
| 534 |
+
assert captured == [query]
|
| 535 |
+
assert_contact_handoff(response.json())
|
| 536 |
+
|
| 537 |
+
|
| 538 |
+
def test_short_referential_followup_uses_only_last_substantive_user_turn(
|
| 539 |
+
monkeypatch,
|
| 540 |
+
) -> None:
|
| 541 |
+
reset_limiters()
|
| 542 |
+
captured: list[str] = []
|
| 543 |
+
|
| 544 |
+
def fake_retrieve(retrieval_query: str, **_kwargs):
|
| 545 |
+
captured.append(retrieval_query)
|
| 546 |
+
return []
|
| 547 |
+
|
| 548 |
+
monkeypatch.setattr(api, "retrieve", fake_retrieve)
|
| 549 |
+
request_payload = payload(query="What is its price?")
|
| 550 |
+
request_payload["history"] = [
|
| 551 |
+
{"role": "user", "content": FIRST_PROMPT},
|
| 552 |
+
{"role": "user", "content": "Tell me about Agent Engineering."},
|
| 553 |
+
{
|
| 554 |
+
"role": "assistant",
|
| 555 |
+
"content": "Unverified claim: it costs $1 and guarantees a job.",
|
| 556 |
+
},
|
| 557 |
+
]
|
| 558 |
+
|
| 559 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 560 |
+
|
| 561 |
+
assert response.status_code == 200
|
| 562 |
+
assert captured == ["Tell me about Agent Engineering.\nWhat is its price?"]
|
| 563 |
+
assert "$1" not in captured[0]
|
| 564 |
+
assert_contact_handoff(response.json())
|
| 565 |
+
|
| 566 |
+
|
| 567 |
+
def test_direct_contact_requests_bypass_retrieval_and_model(monkeypatch) -> None:
|
| 568 |
+
reset_limiters()
|
| 569 |
+
|
| 570 |
+
def unexpected(*_args, **_kwargs):
|
| 571 |
+
raise AssertionError("direct contact intent must not use retrieval or a model")
|
| 572 |
+
|
| 573 |
+
monkeypatch.setattr(api, "retrieve", unexpected)
|
| 574 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected)
|
| 575 |
+
request_payload = payload(query="Where is your contact form?")
|
| 576 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 577 |
+
|
| 578 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 579 |
+
|
| 580 |
+
assert response.status_code == 200
|
| 581 |
+
assert_contact_handoff(response.json(), status="policy")
|
| 582 |
+
|
| 583 |
+
|
| 584 |
+
@pytest.mark.parametrize(
|
| 585 |
+
"query",
|
| 586 |
+
[
|
| 587 |
+
"Does Agent Engineering free preview include lifetime access and require no card?",
|
| 588 |
+
"Does the Full Stack free preview give me a certificate?",
|
| 589 |
+
"Does Building LLMs for Production have a refund guarantee?",
|
| 590 |
+
"How long is Agent Engineering?",
|
| 591 |
+
"How many hours does Python for AI Engineering take?",
|
| 592 |
+
"How many lessons are in LLM Fundamentals?",
|
| 593 |
+
"Does mentorship include a certificate?",
|
| 594 |
+
"Does the Get It All bundle get 50% off?",
|
| 595 |
+
"What is the Get It All bundle price?",
|
| 596 |
+
"Does monthly mentorship have a discount?",
|
| 597 |
+
],
|
| 598 |
+
)
|
| 599 |
+
def test_absent_offer_facts_go_directly_to_contact_form(
|
| 600 |
+
monkeypatch, query: str
|
| 601 |
+
) -> None:
|
| 602 |
+
reset_limiters()
|
| 603 |
+
|
| 604 |
+
def unexpected(*_args, **_kwargs):
|
| 605 |
+
raise AssertionError("model must not run without target-qualified evidence")
|
| 606 |
+
|
| 607 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected)
|
| 608 |
+
request_payload = payload(query=query)
|
| 609 |
+
request_payload["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
| 610 |
+
|
| 611 |
+
response = client.post("/api/helper/chat", json=request_payload, headers=HEADERS)
|
| 612 |
+
|
| 613 |
+
assert response.status_code == 200
|
| 614 |
+
assert_contact_handoff(response.json())
|
| 615 |
+
|
| 616 |
+
|
| 617 |
def test_coupon_answer_is_deterministic_and_does_not_call_model(monkeypatch) -> None:
|
| 618 |
reset_limiters()
|
| 619 |
|
| 620 |
+
def unexpected_generate_answer(
|
| 621 |
+
_prompt: str, _selected: list[dict], **_kwargs
|
| 622 |
+
) -> GroundingResult:
|
| 623 |
raise AssertionError("model should not be called for coupon intent")
|
| 624 |
|
| 625 |
+
monkeypatch.setattr(api.llm, "generate_grounded_answer", unexpected_generate_answer)
|
| 626 |
first = payload(query="Do you have a coupon code?")
|
| 627 |
first["selectedPrompt"] = FIRST_PROMPT
|
| 628 |
first["history"] = [{"role": "user", "content": FIRST_PROMPT}]
|
|
|
|
| 637 |
second_response = client.post("/api/helper/chat", json=second, headers=HEADERS)
|
| 638 |
|
| 639 |
assert first_response.status_code == 200
|
| 640 |
+
assert api.CONTACT_FORM_URL in first_response.json()["answer"]
|
| 641 |
+
assert first_response.json()["status"] == "policy"
|
| 642 |
assert second_response.status_code == 200
|
| 643 |
+
assert api.CONTACT_FORM_URL in second_response.json()["answer"]
|
| 644 |
|
| 645 |
|
| 646 |
def test_rate_limit_is_hard(monkeypatch) -> None:
|
|
|
|
| 652 |
|
| 653 |
monkeypatch.setattr(
|
| 654 |
api.llm,
|
| 655 |
+
"generate_grounded_answer",
|
| 656 |
+
lambda _prompt, selected, **_kwargs: GroundingResult(
|
| 657 |
+
valid=True,
|
| 658 |
+
status="answered",
|
| 659 |
+
answer="Supported answer.",
|
| 660 |
+
cited_chunks=(
|
| 661 |
+
EvidenceChunk(
|
| 662 |
+
chunk_id=str(selected[0]["chunk_id"]),
|
| 663 |
+
title=str(selected[0]["title"]),
|
| 664 |
+
url=str(selected[0]["url"]),
|
| 665 |
+
kind=str(selected[0]["kind"]),
|
| 666 |
+
headings=(),
|
| 667 |
+
text=str(selected[0]["text"]),
|
| 668 |
+
),
|
| 669 |
+
),
|
| 670 |
+
),
|
| 671 |
)
|
| 672 |
|
| 673 |
first = client.post("/api/helper/chat", json=payload(), headers=HEADERS)
|
tests/test_catalog.py
CHANGED
|
@@ -1,5 +1,7 @@
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
|
|
|
|
|
|
| 3 |
from tai_helper.catalog import (
|
| 4 |
allowed_paths_by_host,
|
| 5 |
coupon_followup,
|
|
@@ -7,10 +9,102 @@ from tai_helper.catalog import (
|
|
| 7 |
forced_prompts,
|
| 8 |
in_scope,
|
| 9 |
page_is_allowed,
|
|
|
|
| 10 |
retrieve,
|
| 11 |
sources_from_pages,
|
| 12 |
)
|
| 13 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 14 |
|
| 15 |
def test_forced_prompts_match_expected_options() -> None:
|
| 16 |
assert forced_prompts() == [
|
|
@@ -26,11 +120,17 @@ def test_allowed_public_pages_include_sitemap_urls_and_exclude_private_paths() -
|
|
| 26 |
assert page_is_allowed("https://academy.towardsai.net/courses/agent-engineering")
|
| 27 |
assert page_is_allowed("https://towardsai.net/b2b")
|
| 28 |
assert page_is_allowed("https://www.towardsai.net/b2b")
|
|
|
|
|
|
|
|
|
|
| 29 |
assert not page_is_allowed(
|
| 30 |
"https://academy.towardsai.net/courses/take/agent-engineering/lessons/x"
|
| 31 |
)
|
| 32 |
assert not page_is_allowed("https://academy.towardsai.net/enroll/123")
|
|
|
|
|
|
|
| 33 |
assert "/b2b" in allowed_paths_by_host()["www.towardsai.net"]
|
|
|
|
| 34 |
|
| 35 |
|
| 36 |
def test_retrieval_routes_mentorship_b2b_and_bundle_queries() -> None:
|
|
@@ -38,9 +138,93 @@ def test_retrieval_routes_mentorship_b2b_and_bundle_queries() -> None:
|
|
| 38 |
b2b = sources_from_pages(retrieve("I want training inside my company"))
|
| 39 |
bundle = sources_from_pages(retrieve("What is the best value bundle?"))
|
| 40 |
|
| 41 |
-
assert any(
|
| 42 |
-
|
| 43 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
|
| 45 |
|
| 46 |
def test_scope_and_coupon_detection() -> None:
|
|
|
|
| 1 |
from __future__ import annotations
|
| 2 |
|
| 3 |
+
import pytest
|
| 4 |
+
|
| 5 |
from tai_helper.catalog import (
|
| 6 |
allowed_paths_by_host,
|
| 7 |
coupon_followup,
|
|
|
|
| 9 |
forced_prompts,
|
| 10 |
in_scope,
|
| 11 |
page_is_allowed,
|
| 12 |
+
pages,
|
| 13 |
retrieve,
|
| 14 |
sources_from_pages,
|
| 15 |
)
|
| 16 |
|
| 17 |
+
CANONICAL_OFFER_RETRIEVAL_CASES = (
|
| 18 |
+
(
|
| 19 |
+
"Full Stack AI Engineering production LLM product course",
|
| 20 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 21 |
+
),
|
| 22 |
+
(
|
| 23 |
+
"Agent Engineering production AI agents course",
|
| 24 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 25 |
+
),
|
| 26 |
+
(
|
| 27 |
+
"10-Hour LLM Fundamentals video crash course",
|
| 28 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 29 |
+
),
|
| 30 |
+
(
|
| 31 |
+
"Beginner Python for AI Engineering LLM-native course",
|
| 32 |
+
"https://towardsai.com/academy/python-for-ai-engineering/",
|
| 33 |
+
),
|
| 34 |
+
(
|
| 35 |
+
"Master AI for Work no-code professionals course",
|
| 36 |
+
"https://towardsai.com/academy/ai-for-work/",
|
| 37 |
+
),
|
| 38 |
+
(
|
| 39 |
+
"Building LLMs for Production ebook taught as a course",
|
| 40 |
+
"https://towardsai.com/academy/building-llms-for-production/",
|
| 41 |
+
),
|
| 42 |
+
(
|
| 43 |
+
"Get It All every course one bundle",
|
| 44 |
+
"https://towardsai.com/academy/bundles/get-it-all/",
|
| 45 |
+
),
|
| 46 |
+
(
|
| 47 |
+
"From Non-Coder to AI Engineer course bundle",
|
| 48 |
+
"https://towardsai.com/academy/bundles/from-coding-novice-to-advanced-llm-developer/",
|
| 49 |
+
),
|
| 50 |
+
(
|
| 51 |
+
"From Developer to Advanced AI Engineer course bundle",
|
| 52 |
+
"https://towardsai.com/academy/bundles/10-hour-crash-course-into-llm-developer-expert/",
|
| 53 |
+
),
|
| 54 |
+
(
|
| 55 |
+
"Towards AI Mentorship senior engineers on call",
|
| 56 |
+
"https://towardsai.com/academy/mentorship/",
|
| 57 |
+
),
|
| 58 |
+
(
|
| 59 |
+
"Agent Engineering course preview 7 free lessons",
|
| 60 |
+
"https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 61 |
+
),
|
| 62 |
+
(
|
| 63 |
+
"Full Stack AI Engineering free preview lessons",
|
| 64 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 65 |
+
),
|
| 66 |
+
(
|
| 67 |
+
"Agentic AI engineering webinar",
|
| 68 |
+
"https://towardsai.com/webinars/agentengineering/",
|
| 69 |
+
),
|
| 70 |
+
(
|
| 71 |
+
"Towards AI free Building LLMs book resources and learning toolkit",
|
| 72 |
+
"https://towardsai.com/academy/book/",
|
| 73 |
+
),
|
| 74 |
+
(
|
| 75 |
+
"Towards AI ultimate agents cheatsheet digital download",
|
| 76 |
+
"https://academy.towardsai.net/products/digital_downloads/agents-cheatsheet",
|
| 77 |
+
),
|
| 78 |
+
(
|
| 79 |
+
"Anti-slop framework digital download",
|
| 80 |
+
"https://academy.towardsai.net/products/digital_downloads/anti-slop-framework",
|
| 81 |
+
),
|
| 82 |
+
(
|
| 83 |
+
"Building AI for Production book resources and links",
|
| 84 |
+
"https://towardsai.com/academy/book/",
|
| 85 |
+
),
|
| 86 |
+
(
|
| 87 |
+
"The AI Taste Gap",
|
| 88 |
+
"https://towardsai.com/theaitastegap/",
|
| 89 |
+
),
|
| 90 |
+
(
|
| 91 |
+
"Claude Code Codex one-day agentic developer bootcamp team training",
|
| 92 |
+
"https://towardsai.com/enterprise/agentic-developer-conversion/",
|
| 93 |
+
),
|
| 94 |
+
(
|
| 95 |
+
"Convert software developers into AI engineers enterprise bootcamp",
|
| 96 |
+
"https://towardsai.com/enterprise/software-developer-to-ai-engineer/",
|
| 97 |
+
),
|
| 98 |
+
(
|
| 99 |
+
"enterprise enablement academy",
|
| 100 |
+
"https://towardsai.com/enterpriseenablement/",
|
| 101 |
+
),
|
| 102 |
+
(
|
| 103 |
+
"Custom AI development deployment private equity value creation consulting",
|
| 104 |
+
"https://towardsai.com/valuecreation/",
|
| 105 |
+
),
|
| 106 |
+
)
|
| 107 |
+
|
| 108 |
|
| 109 |
def test_forced_prompts_match_expected_options() -> None:
|
| 110 |
assert forced_prompts() == [
|
|
|
|
| 120 |
assert page_is_allowed("https://academy.towardsai.net/courses/agent-engineering")
|
| 121 |
assert page_is_allowed("https://towardsai.net/b2b")
|
| 122 |
assert page_is_allowed("https://www.towardsai.net/b2b")
|
| 123 |
+
assert page_is_allowed("https://towardsai.com/academy/full-stack-ai-engineering/")
|
| 124 |
+
assert page_is_allowed("https://towardsai.com/academy/mentorship/")
|
| 125 |
+
assert page_is_allowed("https://www.towardsai.com/a-future-public-page/")
|
| 126 |
assert not page_is_allowed(
|
| 127 |
"https://academy.towardsai.net/courses/take/agent-engineering/lessons/x"
|
| 128 |
)
|
| 129 |
assert not page_is_allowed("https://academy.towardsai.net/enroll/123")
|
| 130 |
+
assert not page_is_allowed("https://towardsai.com/wp-admin/edit.php")
|
| 131 |
+
assert not page_is_allowed("https://towardsai.com/academy/?preview=true")
|
| 132 |
assert "/b2b" in allowed_paths_by_host()["www.towardsai.net"]
|
| 133 |
+
assert "/academy/mentorship" in allowed_paths_by_host()["www.towardsai.com"]
|
| 134 |
|
| 135 |
|
| 136 |
def test_retrieval_routes_mentorship_b2b_and_bundle_queries() -> None:
|
|
|
|
| 138 |
b2b = sources_from_pages(retrieve("I want training inside my company"))
|
| 139 |
bundle = sources_from_pages(retrieve("What is the best value bundle?"))
|
| 140 |
|
| 141 |
+
assert any(
|
| 142 |
+
"towardsai.com/academy/mentorship" in source["url"] for source in mentorship
|
| 143 |
+
)
|
| 144 |
+
assert any(source["url"].startswith("https://towardsai.com/") for source in b2b)
|
| 145 |
+
assert any(
|
| 146 |
+
source["url"] == "https://towardsai.com/academy/bundles/get-it-all/"
|
| 147 |
+
for source in bundle
|
| 148 |
+
)
|
| 149 |
+
|
| 150 |
+
|
| 151 |
+
def test_retrieval_routes_specialized_enterprise_queries() -> None:
|
| 152 |
+
coding_agents = sources_from_pages(
|
| 153 |
+
retrieve("We need Claude Code and Codex training for our engineering team")
|
| 154 |
+
)
|
| 155 |
+
developer_conversion = sources_from_pages(
|
| 156 |
+
retrieve("Convert our software developers into AI engineers")
|
| 157 |
+
)
|
| 158 |
+
consulting = sources_from_pages(
|
| 159 |
+
retrieve("We need AI deployment and value creation consulting")
|
| 160 |
+
)
|
| 161 |
+
|
| 162 |
+
assert coding_agents[0]["url"].endswith("/agentic-developer-conversion/")
|
| 163 |
+
assert developer_conversion[0]["url"].endswith(
|
| 164 |
+
"/software-developer-to-ai-engineer/"
|
| 165 |
+
)
|
| 166 |
+
assert consulting[0]["url"] == "https://towardsai.com/valuecreation/"
|
| 167 |
+
|
| 168 |
+
|
| 169 |
+
@pytest.mark.parametrize(
|
| 170 |
+
("query", "expected_url"),
|
| 171 |
+
CANONICAL_OFFER_RETRIEVAL_CASES,
|
| 172 |
+
ids=[
|
| 173 |
+
expected_url.rstrip("/").rsplit("/", 1)[-1]
|
| 174 |
+
for _, expected_url in CANONICAL_OFFER_RETRIEVAL_CASES
|
| 175 |
+
],
|
| 176 |
+
)
|
| 177 |
+
def test_every_current_offer_routes_to_its_canonical_page(
|
| 178 |
+
query: str, expected_url: str
|
| 179 |
+
) -> None:
|
| 180 |
+
selected = retrieve(query, limit=1)
|
| 181 |
+
|
| 182 |
+
assert selected
|
| 183 |
+
assert selected[0]["url"] == expected_url
|
| 184 |
+
|
| 185 |
+
|
| 186 |
+
def test_current_mentorship_offer_replaces_stale_membership_claim() -> None:
|
| 187 |
+
mentorship_pages = [
|
| 188 |
+
page
|
| 189 |
+
for page in pages()
|
| 190 |
+
if page.get("url") == "https://towardsai.com/academy/mentorship/"
|
| 191 |
+
]
|
| 192 |
+
|
| 193 |
+
assert len(mentorship_pages) == 1
|
| 194 |
+
mentorship = mentorship_pages[0]
|
| 195 |
+
searchable_text = " ".join(
|
| 196 |
+
[
|
| 197 |
+
str(mentorship.get("meta_description", "")),
|
| 198 |
+
" ".join(str(item) for item in mentorship.get("headings", [])),
|
| 199 |
+
str(mentorship.get("text", "")),
|
| 200 |
+
]
|
| 201 |
+
)
|
| 202 |
+
assert "10-Hour LLM Fundamentals" in searchable_text
|
| 203 |
+
assert "Two courses included" not in searchable_text
|
| 204 |
+
chunks = mentorship.get("chunks", [])
|
| 205 |
+
assert chunks
|
| 206 |
+
assert all(len(str(chunk.get("text", ""))) <= 2400 for chunk in chunks)
|
| 207 |
+
offer_chunks = [
|
| 208 |
+
str(chunk.get("text", ""))
|
| 209 |
+
for chunk in chunks
|
| 210 |
+
if chunk.get("heading") == "The curriculum comes with the team"
|
| 211 |
+
]
|
| 212 |
+
assert len(offer_chunks) == 1
|
| 213 |
+
offer_table = offer_chunks[0]
|
| 214 |
+
assert offer_table.count("Included from day one") == 1
|
| 215 |
+
assert (
|
| 216 |
+
"10-Hour LLM Fundamentals video course Five in-depth 2-hour video "
|
| 217 |
+
"sessions, from a basic prompt to a full production rollout $199 "
|
| 218 |
+
"Included from day one"
|
| 219 |
+
) in offer_table
|
| 220 |
+
assert (
|
| 221 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 222 |
+
"$349–499 each 25% off, always"
|
| 223 |
+
) in offer_table
|
| 224 |
+
assert not any(
|
| 225 |
+
page.get("url") == "https://towardsai.com/academy/membership/"
|
| 226 |
+
for page in pages()
|
| 227 |
+
)
|
| 228 |
|
| 229 |
|
| 230 |
def test_scope_and_coupon_detection() -> None:
|
tests/test_catalog_builder.py
ADDED
|
@@ -0,0 +1,487 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import requests
|
| 4 |
+
|
| 5 |
+
from scripts.build_towardsai_com_catalog import (
|
| 6 |
+
ACADEMY_EXCLUSIONS,
|
| 7 |
+
ACADEMY_HOSTS,
|
| 8 |
+
ACADEMY_SITEMAP_URL,
|
| 9 |
+
COM_HOSTS,
|
| 10 |
+
PageParser,
|
| 11 |
+
build_catalog,
|
| 12 |
+
build_catalogs,
|
| 13 |
+
build_chunks,
|
| 14 |
+
discover_sitemap_entries,
|
| 15 |
+
fetch_page,
|
| 16 |
+
)
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
class FakeResponse:
|
| 20 |
+
def __init__(
|
| 21 |
+
self,
|
| 22 |
+
url: str,
|
| 23 |
+
text: str,
|
| 24 |
+
*,
|
| 25 |
+
status_code: int = 200,
|
| 26 |
+
) -> None:
|
| 27 |
+
self.url = url
|
| 28 |
+
self.text = text
|
| 29 |
+
self.status_code = status_code
|
| 30 |
+
|
| 31 |
+
def raise_for_status(self) -> None:
|
| 32 |
+
if self.status_code >= 400:
|
| 33 |
+
raise requests.HTTPError(f"HTTP {self.status_code} for {self.url}")
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class FakeSession:
|
| 37 |
+
def __init__(self, routes: dict[str, FakeResponse]) -> None:
|
| 38 |
+
self.routes = routes
|
| 39 |
+
self.headers: dict[str, str] = {}
|
| 40 |
+
self.calls: list[str] = []
|
| 41 |
+
|
| 42 |
+
def get(self, url: str, *, timeout: int) -> FakeResponse:
|
| 43 |
+
assert timeout == 30
|
| 44 |
+
self.calls.append(url)
|
| 45 |
+
if url not in self.routes:
|
| 46 |
+
raise AssertionError(f"unexpected network request: {url}")
|
| 47 |
+
return self.routes[url]
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
def sitemap(*urls: str) -> str:
|
| 51 |
+
rows = "".join(f"<url><loc>{url}</loc></url>" for url in urls)
|
| 52 |
+
return (
|
| 53 |
+
'<?xml version="1.0"?>'
|
| 54 |
+
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">'
|
| 55 |
+
f"{rows}</urlset>"
|
| 56 |
+
)
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def page_html(title: str, body: str, *, canonical: str = "") -> str:
|
| 60 |
+
canonical_tag = (
|
| 61 |
+
f'<link href="{canonical}" rel="alternate canonical">' if canonical else ""
|
| 62 |
+
)
|
| 63 |
+
return f"""
|
| 64 |
+
<html>
|
| 65 |
+
<head>
|
| 66 |
+
<title>{title}</title>
|
| 67 |
+
<meta name="description" content="A fetched description">
|
| 68 |
+
{canonical_tag}
|
| 69 |
+
</head>
|
| 70 |
+
<body>
|
| 71 |
+
<nav>Navigation misinformation</nav>
|
| 72 |
+
<main><h1>{title}</h1><p>{body}</p></main>
|
| 73 |
+
<footer>Footer misinformation</footer>
|
| 74 |
+
<script>Script misinformation</script>
|
| 75 |
+
<style>.misinformation {{ display: block }}</style>
|
| 76 |
+
</body>
|
| 77 |
+
</html>
|
| 78 |
+
"""
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def test_parser_suppresses_chrome_and_extracts_canonical_hints() -> None:
|
| 82 |
+
parser = PageParser("https://towardsai.com/original/")
|
| 83 |
+
parser.feed(
|
| 84 |
+
"""
|
| 85 |
+
<html><head>
|
| 86 |
+
<title>Course title</title>
|
| 87 |
+
<link rel="canonical" href="/academy/course/">
|
| 88 |
+
<meta http-equiv="Refresh" content="0; URL='/moved/'">
|
| 89 |
+
</head><body>
|
| 90 |
+
<header class="header"><a>Skip to main content</a><button>Toggle menu</button></header>
|
| 91 |
+
<header class="ta-hero"><h1>Current offer</h1><p>60+ Hrs</p></header>
|
| 92 |
+
<div class="ta-announce">New: Towards AI Mentorship</div>
|
| 93 |
+
<div class="navbar"><p>Bad navigation claim</p></div>
|
| 94 |
+
<div class="ta-mobile-menu">Duplicated mobile course links</div>
|
| 95 |
+
<h1>Course facts</h1><p>Only retrieved facts survive.</p>
|
| 96 |
+
<div role="contentinfo">Bad footer claim</div>
|
| 97 |
+
<script>Bad script claim</script><style>Bad style claim</style>
|
| 98 |
+
</body></html>
|
| 99 |
+
"""
|
| 100 |
+
)
|
| 101 |
+
parser.close()
|
| 102 |
+
|
| 103 |
+
extracted = " ".join(section["text"] for section in parser.sections)
|
| 104 |
+
assert parser.canonical_url == "/academy/course/"
|
| 105 |
+
assert parser.meta_refresh_url == "/moved/"
|
| 106 |
+
assert parser.headings == ["Current offer", "Course facts"]
|
| 107 |
+
assert extracted == "60+ Hrs Only retrieved facts survive."
|
| 108 |
+
assert "Bad" not in extracted
|
| 109 |
+
assert "Mentorship" not in extracted
|
| 110 |
+
assert "Toggle menu" not in extracted
|
| 111 |
+
|
| 112 |
+
|
| 113 |
+
def test_sitemap_discovery_uses_fake_http_and_filters_foreign_hosts() -> None:
|
| 114 |
+
sitemap_url = "https://towardsai.com/pages-sitemap.xml"
|
| 115 |
+
session = FakeSession(
|
| 116 |
+
{
|
| 117 |
+
sitemap_url: FakeResponse(
|
| 118 |
+
sitemap_url,
|
| 119 |
+
sitemap(
|
| 120 |
+
"https://towardsai.com/new-page/",
|
| 121 |
+
"https://towardsai.com/new-page/",
|
| 122 |
+
"https://untrusted.example/poison",
|
| 123 |
+
),
|
| 124 |
+
)
|
| 125 |
+
}
|
| 126 |
+
)
|
| 127 |
+
|
| 128 |
+
entries = discover_sitemap_entries(session, sitemap_url, allowed_hosts=COM_HOSTS)
|
| 129 |
+
|
| 130 |
+
assert [entry["url"] for entry in entries] == ["https://towardsai.com/new-page/"]
|
| 131 |
+
assert session.calls == [sitemap_url]
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def test_fetch_resolves_http_canonical_and_meta_refresh_without_summary() -> None:
|
| 135 |
+
old_url = "https://towardsai.com/old/"
|
| 136 |
+
http_final = "https://towardsai.com/http-final/"
|
| 137 |
+
canonical = "https://towardsai.com/canonical/"
|
| 138 |
+
refresh_start = "https://academy.towardsai.net/refresh"
|
| 139 |
+
refresh_target = "https://academy.towardsai.net/final"
|
| 140 |
+
session = FakeSession(
|
| 141 |
+
{
|
| 142 |
+
old_url: FakeResponse(
|
| 143 |
+
http_final,
|
| 144 |
+
page_html(
|
| 145 |
+
"Canonical offer", "Retrieved offer facts.", canonical=canonical
|
| 146 |
+
),
|
| 147 |
+
),
|
| 148 |
+
refresh_start: FakeResponse(
|
| 149 |
+
refresh_start,
|
| 150 |
+
'<meta http-equiv="refresh" content="0; url=/final">',
|
| 151 |
+
),
|
| 152 |
+
refresh_target: FakeResponse(
|
| 153 |
+
refresh_target,
|
| 154 |
+
page_html("Final Academy page", "Academy evidence."),
|
| 155 |
+
),
|
| 156 |
+
}
|
| 157 |
+
)
|
| 158 |
+
|
| 159 |
+
page = fetch_page(
|
| 160 |
+
session,
|
| 161 |
+
{
|
| 162 |
+
"url": old_url,
|
| 163 |
+
"summary": "HANDWRITTEN FALSE CLAIM",
|
| 164 |
+
"kind": "course",
|
| 165 |
+
},
|
| 166 |
+
allowed_hosts=COM_HOSTS,
|
| 167 |
+
fetched_at="2026-08-08T00:00:00+00:00",
|
| 168 |
+
)
|
| 169 |
+
refreshed = fetch_page(
|
| 170 |
+
session,
|
| 171 |
+
{"url": refresh_start},
|
| 172 |
+
authority="official_academy",
|
| 173 |
+
allowed_hosts=ACADEMY_HOSTS,
|
| 174 |
+
fetched_at="2026-08-08T00:00:00+00:00",
|
| 175 |
+
)
|
| 176 |
+
|
| 177 |
+
assert page["discovered_url"] == old_url
|
| 178 |
+
assert page["url"] == canonical
|
| 179 |
+
assert page["canonical_url"] == canonical
|
| 180 |
+
assert page["redirect_chain"] == [http_final]
|
| 181 |
+
assert "Retrieved offer facts." in page["text"]
|
| 182 |
+
assert "HANDWRITTEN FALSE CLAIM" not in page["text"]
|
| 183 |
+
assert "Navigation misinformation" not in page["text"]
|
| 184 |
+
assert refreshed["canonical_url"] == refresh_target
|
| 185 |
+
assert refreshed["redirect_chain"] == [refresh_start, refresh_target]
|
| 186 |
+
|
| 187 |
+
|
| 188 |
+
def test_chunks_are_heading_aware_bounded_and_stable() -> None:
|
| 189 |
+
sections = [
|
| 190 |
+
{
|
| 191 |
+
"heading": "Mentorship access",
|
| 192 |
+
"text": "evidence " * 350,
|
| 193 |
+
}
|
| 194 |
+
]
|
| 195 |
+
changed_sections = [
|
| 196 |
+
{
|
| 197 |
+
"heading": "Mentorship access",
|
| 198 |
+
"text": ("evidence " * 349) + "updated",
|
| 199 |
+
}
|
| 200 |
+
]
|
| 201 |
+
|
| 202 |
+
first = build_chunks(sections, "https://towardsai.com/academy/mentorship/")
|
| 203 |
+
second = build_chunks(sections, "https://towardsai.com/academy/mentorship/")
|
| 204 |
+
changed = build_chunks(
|
| 205 |
+
changed_sections, "https://towardsai.com/academy/mentorship/"
|
| 206 |
+
)
|
| 207 |
+
|
| 208 |
+
assert len(first) == 2
|
| 209 |
+
assert [chunk["chunk_id"] for chunk in first] == [
|
| 210 |
+
chunk["chunk_id"] for chunk in second
|
| 211 |
+
]
|
| 212 |
+
assert [chunk["chunk_id"] for chunk in first] == [
|
| 213 |
+
chunk["chunk_id"] for chunk in changed
|
| 214 |
+
]
|
| 215 |
+
assert all(chunk["heading"] == "Mentorship access" for chunk in first)
|
| 216 |
+
assert all(1_200 <= len(chunk["text"]) <= 1_800 for chunk in first)
|
| 217 |
+
|
| 218 |
+
|
| 219 |
+
def test_table_rows_become_atomic_evidence_spans() -> None:
|
| 220 |
+
parser = PageParser("https://towardsai.com/academy/mentorship/")
|
| 221 |
+
parser.feed(
|
| 222 |
+
"""
|
| 223 |
+
<main>
|
| 224 |
+
<h2>The curriculum comes with the team</h2>
|
| 225 |
+
<table><tbody>
|
| 226 |
+
<tr><th>10-Hour LLM Fundamentals course</th><td>$199</td>
|
| 227 |
+
<td>Included from day one</td></tr>
|
| 228 |
+
<tr><th>Full Stack AI Engineering</th><td>$349</td>
|
| 229 |
+
<td>25% off, always</td></tr>
|
| 230 |
+
</tbody></table>
|
| 231 |
+
</main>
|
| 232 |
+
"""
|
| 233 |
+
)
|
| 234 |
+
parser.close()
|
| 235 |
+
|
| 236 |
+
built = build_chunks(parser.sections, parser.base_url)
|
| 237 |
+
spans = {span["text"] for chunk in built for span in chunk["evidence_spans"]}
|
| 238 |
+
|
| 239 |
+
assert "10-Hour LLM Fundamentals course $199 Included from day one" in spans
|
| 240 |
+
assert "Full Stack AI Engineering $349 25% off, always" in spans
|
| 241 |
+
assert "Included from day one" not in spans
|
| 242 |
+
assert "25% off, always" not in spans
|
| 243 |
+
|
| 244 |
+
|
| 245 |
+
def test_accordion_uses_only_the_answer_as_an_atomic_evidence_span() -> None:
|
| 246 |
+
built = build_chunks(
|
| 247 |
+
[
|
| 248 |
+
{
|
| 249 |
+
"heading": "Frequently asked questions",
|
| 250 |
+
"text": (
|
| 251 |
+
"How long does this course take to complete? + "
|
| 252 |
+
"It is self-paced; the average completion time is 10 hours "
|
| 253 |
+
"across five 2-hour sessions."
|
| 254 |
+
),
|
| 255 |
+
"spans": [
|
| 256 |
+
(
|
| 257 |
+
"How long does this course take to complete? + "
|
| 258 |
+
"It is self-paced; the average completion time is 10 hours "
|
| 259 |
+
"across five 2-hour sessions."
|
| 260 |
+
)
|
| 261 |
+
],
|
| 262 |
+
}
|
| 263 |
+
],
|
| 264 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 265 |
+
)
|
| 266 |
+
spans = {span["text"] for chunk in built for span in chunk["evidence_spans"]}
|
| 267 |
+
|
| 268 |
+
assert "How long does this course take to complete?" not in spans
|
| 269 |
+
assert (
|
| 270 |
+
"It is self-paced; the average completion time is 10 hours across five "
|
| 271 |
+
"2-hour sessions."
|
| 272 |
+
) in spans
|
| 273 |
+
assert not any("? +" in span for span in spans)
|
| 274 |
+
|
| 275 |
+
|
| 276 |
+
def test_build_both_catalogs_records_exclusions_manual_entries_and_identity() -> None:
|
| 277 |
+
fetched_at = "2026-08-08T12:00:00+00:00"
|
| 278 |
+
com_sitemap = "https://towardsai.com/pages-sitemap.xml"
|
| 279 |
+
academy_sitemap = "https://academy.towardsai.net/sitemap.xml"
|
| 280 |
+
mentorship = "https://towardsai.com/academy/mentorship/"
|
| 281 |
+
academy_mentorship = "https://academy.towardsai.net/bundles/tai-mentorship"
|
| 282 |
+
placeholder = "https://academy.towardsai.net/pages/webinar"
|
| 283 |
+
session = FakeSession(
|
| 284 |
+
{
|
| 285 |
+
com_sitemap: FakeResponse(com_sitemap, sitemap(mentorship)),
|
| 286 |
+
academy_sitemap: FakeResponse(
|
| 287 |
+
academy_sitemap, sitemap(academy_mentorship, placeholder)
|
| 288 |
+
),
|
| 289 |
+
mentorship: FakeResponse(
|
| 290 |
+
mentorship,
|
| 291 |
+
page_html(
|
| 292 |
+
"Mentorship",
|
| 293 |
+
"The retrieved page defines the included course access.",
|
| 294 |
+
),
|
| 295 |
+
),
|
| 296 |
+
academy_mentorship: FakeResponse(
|
| 297 |
+
academy_mentorship,
|
| 298 |
+
page_html("Mentorship checkout", "Thinkific purchase facts."),
|
| 299 |
+
),
|
| 300 |
+
placeholder: FakeResponse(
|
| 301 |
+
placeholder,
|
| 302 |
+
page_html("Webinar", "Add a concise subheading about your product."),
|
| 303 |
+
),
|
| 304 |
+
}
|
| 305 |
+
)
|
| 306 |
+
|
| 307 |
+
com, academy = build_catalogs(session=session, fetched_at=fetched_at)
|
| 308 |
+
|
| 309 |
+
assert [page["discovered_url"] for page in com["pages"]] == [mentorship]
|
| 310 |
+
assert len(academy["pages"]) == 4
|
| 311 |
+
com_offer = com["pages"][0]
|
| 312 |
+
academy_offer = next(
|
| 313 |
+
page
|
| 314 |
+
for page in academy["pages"]
|
| 315 |
+
if page["discovered_url"] == academy_mentorship
|
| 316 |
+
)
|
| 317 |
+
excluded = next(
|
| 318 |
+
page for page in academy["pages"] if page["discovered_url"] == placeholder
|
| 319 |
+
)
|
| 320 |
+
manual = [
|
| 321 |
+
page for page in academy["pages"] if page["authority"] == "curated_external"
|
| 322 |
+
]
|
| 323 |
+
|
| 324 |
+
assert com_offer["offer_id"] == academy_offer["offer_id"] == "mentorship"
|
| 325 |
+
assert com_offer["entity_id"] == academy_offer["entity_id"]
|
| 326 |
+
assert excluded["status"] == "excluded"
|
| 327 |
+
assert excluded["retrieval_eligible"] is False
|
| 328 |
+
assert excluded["excluded_reason"] == "unfinished template content"
|
| 329 |
+
assert placeholder in session.calls # excluded URLs are still fetched and hashed
|
| 330 |
+
assert len(manual) == 2
|
| 331 |
+
assert all(page["status"] == "excluded" for page in manual)
|
| 332 |
+
assert all(page["retrieval_eligible"] is False for page in manual)
|
| 333 |
+
assert all(page["chunks"] == [] for page in manual)
|
| 334 |
+
|
| 335 |
+
required = {
|
| 336 |
+
"canonical_url",
|
| 337 |
+
"fetched_at",
|
| 338 |
+
"content_sha256",
|
| 339 |
+
"content_hash",
|
| 340 |
+
"evidence_hash",
|
| 341 |
+
"authority",
|
| 342 |
+
"status",
|
| 343 |
+
"retrieval_eligible",
|
| 344 |
+
"chunks",
|
| 345 |
+
"entity_id",
|
| 346 |
+
"offer_id",
|
| 347 |
+
}
|
| 348 |
+
assert all(required <= page.keys() for page in [*com["pages"], *academy["pages"]])
|
| 349 |
+
assert all(page["fetched_at"] == fetched_at for page in com["pages"])
|
| 350 |
+
|
| 351 |
+
|
| 352 |
+
def test_failed_page_is_recorded_but_never_retrieval_eligible() -> None:
|
| 353 |
+
sitemap_url = "https://towardsai.com/pages-sitemap.xml"
|
| 354 |
+
broken_url = "https://towardsai.com/broken/"
|
| 355 |
+
session = FakeSession(
|
| 356 |
+
{
|
| 357 |
+
sitemap_url: FakeResponse(sitemap_url, sitemap(broken_url)),
|
| 358 |
+
broken_url: FakeResponse(broken_url, "upstream failed", status_code=503),
|
| 359 |
+
}
|
| 360 |
+
)
|
| 361 |
+
|
| 362 |
+
catalog = build_catalog(
|
| 363 |
+
session=session,
|
| 364 |
+
sitemap_url=sitemap_url,
|
| 365 |
+
fetched_at="2026-08-08T00:00:00+00:00",
|
| 366 |
+
)
|
| 367 |
+
|
| 368 |
+
assert len(catalog["pages"]) == 1
|
| 369 |
+
page = catalog["pages"][0]
|
| 370 |
+
assert page["status"] == "fetch_error"
|
| 371 |
+
assert page["retrieval_eligible"] is False
|
| 372 |
+
assert page["chunks"] == []
|
| 373 |
+
assert page["content_sha256"] == page["content_hash"]
|
| 374 |
+
assert len(page["evidence_hash"]) == 64
|
| 375 |
+
assert catalog["status_counts"] == {"fetch_error": 1}
|
| 376 |
+
|
| 377 |
+
|
| 378 |
+
def test_stale_academy_summaries_are_fetched_but_excluded_from_retrieval() -> None:
|
| 379 |
+
stale_paths = (
|
| 380 |
+
"/collections",
|
| 381 |
+
"/collections/products",
|
| 382 |
+
"/collections/developers",
|
| 383 |
+
"/collections/professionals",
|
| 384 |
+
"/pages/free-resources",
|
| 385 |
+
)
|
| 386 |
+
urls = tuple(f"https://academy.towardsai.net{path}" for path in stale_paths)
|
| 387 |
+
routes = {
|
| 388 |
+
ACADEMY_SITEMAP_URL: FakeResponse(
|
| 389 |
+
ACADEMY_SITEMAP_URL,
|
| 390 |
+
sitemap(*urls),
|
| 391 |
+
)
|
| 392 |
+
}
|
| 393 |
+
routes.update(
|
| 394 |
+
{
|
| 395 |
+
url: FakeResponse(
|
| 396 |
+
url,
|
| 397 |
+
page_html("Legacy Academy summary", f"Stale content for {path}."),
|
| 398 |
+
)
|
| 399 |
+
for url, path in zip(urls, stale_paths, strict=True)
|
| 400 |
+
}
|
| 401 |
+
)
|
| 402 |
+
session = FakeSession(routes)
|
| 403 |
+
|
| 404 |
+
catalog = build_catalog(
|
| 405 |
+
session=session,
|
| 406 |
+
sitemap_url=ACADEMY_SITEMAP_URL,
|
| 407 |
+
authority="official_academy",
|
| 408 |
+
allowed_hosts=ACADEMY_HOSTS,
|
| 409 |
+
fetched_at="2026-08-08T00:00:00+00:00",
|
| 410 |
+
)
|
| 411 |
+
|
| 412 |
+
pages_by_path = {page["path"]: page for page in catalog["pages"]}
|
| 413 |
+
assert set(pages_by_path) == set(stale_paths)
|
| 414 |
+
for path in stale_paths:
|
| 415 |
+
page = pages_by_path[path]
|
| 416 |
+
assert page["status"] == "excluded"
|
| 417 |
+
assert page["retrieval_eligible"] is False
|
| 418 |
+
assert page["excluded_reason"] == ACADEMY_EXCLUSIONS[path]
|
| 419 |
+
assert page["text"]
|
| 420 |
+
assert page["content_hash"] != ""
|
| 421 |
+
assert f"https://academy.towardsai.net{path}" in session.calls
|
| 422 |
+
|
| 423 |
+
|
| 424 |
+
def test_parser_excludes_reviews_comparisons_and_simulated_conversations() -> None:
|
| 425 |
+
parser = PageParser("https://towardsai.com/academy/llm-primer/")
|
| 426 |
+
parser.feed(
|
| 427 |
+
"""
|
| 428 |
+
<main>
|
| 429 |
+
<h2>Official duration</h2><p>Average completion time is 10 hours.</p>
|
| 430 |
+
<article class="ta-review"><p>The course actually has 12 hours.</p></article>
|
| 431 |
+
<div class="ta-ment-vswrap"><p>A competitor costs $500.</p></div>
|
| 432 |
+
<p class="ta-ment-vsnote">A single mentor costs $150 monthly.</p>
|
| 433 |
+
<div class="ta-ment-thread"><p>An example user says guaranteed.</p></div>
|
| 434 |
+
</main>
|
| 435 |
+
"""
|
| 436 |
+
)
|
| 437 |
+
parser.close()
|
| 438 |
+
text = " ".join(section["text"] for section in parser.sections)
|
| 439 |
+
|
| 440 |
+
assert "Average completion time is 10 hours" in text
|
| 441 |
+
assert "12 hours" not in text
|
| 442 |
+
assert "competitor costs" not in text
|
| 443 |
+
assert "single mentor costs" not in text
|
| 444 |
+
assert "example user" not in text
|
| 445 |
+
|
| 446 |
+
|
| 447 |
+
def test_parser_excludes_paid_course_outcomes_from_preview_evidence() -> None:
|
| 448 |
+
parser = PageParser(
|
| 449 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/"
|
| 450 |
+
)
|
| 451 |
+
parser.feed(
|
| 452 |
+
"""
|
| 453 |
+
<main>
|
| 454 |
+
<section id="preview"><h2>Free preview</h2><p>No card required.</p></section>
|
| 455 |
+
<section id="outcomes">
|
| 456 |
+
<h2>The full course outcomes</h2>
|
| 457 |
+
<p>A certification that unlocks six-figure roles.</p>
|
| 458 |
+
</section>
|
| 459 |
+
<section id="cta"><p>Buy the paid course.</p></section>
|
| 460 |
+
</main>
|
| 461 |
+
"""
|
| 462 |
+
)
|
| 463 |
+
parser.close()
|
| 464 |
+
text = " ".join(section["text"] for section in parser.sections)
|
| 465 |
+
|
| 466 |
+
assert "No card required" in text
|
| 467 |
+
assert "certification" not in text
|
| 468 |
+
assert "paid course" not in text
|
| 469 |
+
|
| 470 |
+
|
| 471 |
+
def test_mentorship_comparison_keeps_own_features_but_drops_competitors() -> None:
|
| 472 |
+
parser = PageParser("https://towardsai.com/academy/mentorship/")
|
| 473 |
+
parser.feed(
|
| 474 |
+
"""
|
| 475 |
+
<main><section id="compare">
|
| 476 |
+
<div class="inc"><p>Resume & project reviews in 48–72h</p></div>
|
| 477 |
+
<div class="board"><p>A single mentor costs $150/month.</p></div>
|
| 478 |
+
<p class="sum">Bought separately: $2,000+ a month.</p>
|
| 479 |
+
</section></main>
|
| 480 |
+
"""
|
| 481 |
+
)
|
| 482 |
+
parser.close()
|
| 483 |
+
text = " ".join(section["text"] for section in parser.sections)
|
| 484 |
+
|
| 485 |
+
assert "Resume & project reviews in 48–72h" in text
|
| 486 |
+
assert "single mentor" not in text
|
| 487 |
+
assert "Bought separately" not in text
|
tests/test_grounding.py
ADDED
|
@@ -0,0 +1,866 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import json
|
| 4 |
+
|
| 5 |
+
import pytest
|
| 6 |
+
|
| 7 |
+
from tai_helper import llm
|
| 8 |
+
|
| 9 |
+
MENTORSHIP_PAGE = {
|
| 10 |
+
"title": "Towards AI Mentorship",
|
| 11 |
+
"kind": "mentorship",
|
| 12 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 13 |
+
"headings": ["Everything Inside the Mentorship"],
|
| 14 |
+
"text": (
|
| 15 |
+
"The Mentorship program includes one course: the 10-Hour LLM "
|
| 16 |
+
"Fundamentals course. Members also receive two live sessions every week."
|
| 17 |
+
),
|
| 18 |
+
}
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def _model_json(*, text: str, quote: str, chunk_id: str | None = None) -> str:
|
| 22 |
+
return json.dumps(
|
| 23 |
+
{
|
| 24 |
+
"status": "answered",
|
| 25 |
+
"claims": [
|
| 26 |
+
{
|
| 27 |
+
"text": text,
|
| 28 |
+
"chunk_id": chunk_id or llm.chunk_id_for_page(MENTORSHIP_PAGE),
|
| 29 |
+
"quote": quote,
|
| 30 |
+
}
|
| 31 |
+
],
|
| 32 |
+
}
|
| 33 |
+
)
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def _with_spans(page: dict, *spans: str) -> dict:
|
| 37 |
+
return {
|
| 38 |
+
**page,
|
| 39 |
+
"evidence_spans": [
|
| 40 |
+
{"span_id": f"{page['chunk_id']}:span-{index}", "text": span}
|
| 41 |
+
for index, span in enumerate(spans)
|
| 42 |
+
],
|
| 43 |
+
}
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
def test_prompt_uses_stable_chunk_ids_and_marks_notes_as_non_evidence() -> None:
|
| 47 |
+
prompt = llm.build_prompt(
|
| 48 |
+
query="Does mentorship include a course?",
|
| 49 |
+
selected_prompt="I want to find mentors",
|
| 50 |
+
current_url="https://towardsai.com/academy/mentorship/",
|
| 51 |
+
page_title="Towards AI Mentorship",
|
| 52 |
+
history=[("assistant", "Earlier unverified answer: two courses")],
|
| 53 |
+
selected_pages=[MENTORSHIP_PAGE],
|
| 54 |
+
)
|
| 55 |
+
chunk_id = llm.chunk_id_for_page(MENTORSHIP_PAGE)
|
| 56 |
+
|
| 57 |
+
assert llm.chunk_id_for_page(dict(MENTORSHIP_PAGE)) == chunk_id
|
| 58 |
+
assert f'<EVIDENCE_CHUNK chunk_id="{chunk_id}">' in prompt
|
| 59 |
+
assert "<NON_EVIDENCE_ROUTING_NOTES>" in prompt
|
| 60 |
+
assert "must not be cited" in prompt
|
| 61 |
+
assert "visitor-provided and NOT evidence" in prompt
|
| 62 |
+
assert "return not_found" in prompt
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
def test_valid_mentorship_claim_returns_only_validated_text_and_cited_chunk() -> None:
|
| 66 |
+
sentence = (
|
| 67 |
+
"The Mentorship program includes one course: the 10-Hour LLM "
|
| 68 |
+
"Fundamentals course."
|
| 69 |
+
)
|
| 70 |
+
raw = llm.LLMResult(
|
| 71 |
+
answer=_model_json(text=sentence, quote=sentence),
|
| 72 |
+
usage={"provider": "deepseek"},
|
| 73 |
+
latency_ms=42,
|
| 74 |
+
)
|
| 75 |
+
|
| 76 |
+
result = llm.validate_grounded_result(raw, [MENTORSHIP_PAGE])
|
| 77 |
+
|
| 78 |
+
assert result.valid
|
| 79 |
+
assert result.is_answered
|
| 80 |
+
assert result.status == "answered"
|
| 81 |
+
assert result.answer == sentence
|
| 82 |
+
assert result.cited_chunk_ids == (llm.chunk_id_for_page(MENTORSHIP_PAGE),)
|
| 83 |
+
assert result.cited_chunks[0].url == MENTORSHIP_PAGE["url"]
|
| 84 |
+
assert result.usage == {"provider": "deepseek"}
|
| 85 |
+
assert result.latency_ms == 42
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
def test_valid_unpunctuated_atomic_span_gets_only_a_server_added_period() -> None:
|
| 89 |
+
quote = "10-Hour LLM Fundamentals course $199 Included from day one"
|
| 90 |
+
page = _with_spans(
|
| 91 |
+
{
|
| 92 |
+
**MENTORSHIP_PAGE,
|
| 93 |
+
"chunk_id": "mentorship-row",
|
| 94 |
+
"text": quote,
|
| 95 |
+
},
|
| 96 |
+
quote,
|
| 97 |
+
)
|
| 98 |
+
result = llm.validate_grounded_result(
|
| 99 |
+
_model_json(text=quote, quote=quote, chunk_id="mentorship-row"),
|
| 100 |
+
[page],
|
| 101 |
+
)
|
| 102 |
+
|
| 103 |
+
assert result.valid
|
| 104 |
+
assert result.answer == f"{quote}."
|
| 105 |
+
assert result.claims[0].quote == quote
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
def test_mentorship_table_supports_two_separate_correcting_claims() -> None:
|
| 109 |
+
table_text = (
|
| 110 |
+
"What Full price As a member 10-Hour LLM Fundamentals video course "
|
| 111 |
+
"Five in-depth 2-hour video sessions $199 Included from day one "
|
| 112 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 113 |
+
"$349–499 each 25% off, always"
|
| 114 |
+
)
|
| 115 |
+
included_quote = (
|
| 116 |
+
"10-Hour LLM Fundamentals video course Five in-depth 2-hour video "
|
| 117 |
+
"sessions $199 Included from day one"
|
| 118 |
+
)
|
| 119 |
+
discount_quote = (
|
| 120 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 121 |
+
"$349–499 each 25% off, always"
|
| 122 |
+
)
|
| 123 |
+
table_page = _with_spans(
|
| 124 |
+
{
|
| 125 |
+
"chunk_id": "mentorship-curriculum-table",
|
| 126 |
+
"title": "Towards AI Mentorship",
|
| 127 |
+
"kind": "mentorship",
|
| 128 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 129 |
+
"headings": ["The curriculum comes with the team"],
|
| 130 |
+
"text": table_text,
|
| 131 |
+
},
|
| 132 |
+
included_quote,
|
| 133 |
+
discount_quote,
|
| 134 |
+
)
|
| 135 |
+
raw = json.dumps(
|
| 136 |
+
{
|
| 137 |
+
"status": "answered",
|
| 138 |
+
"claims": [
|
| 139 |
+
{
|
| 140 |
+
"text": f"{included_quote}.",
|
| 141 |
+
"chunk_id": "mentorship-curriculum-table",
|
| 142 |
+
"quote": included_quote,
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"text": f"{discount_quote}.",
|
| 146 |
+
"chunk_id": "mentorship-curriculum-table",
|
| 147 |
+
"quote": discount_quote,
|
| 148 |
+
},
|
| 149 |
+
],
|
| 150 |
+
}
|
| 151 |
+
)
|
| 152 |
+
|
| 153 |
+
result = llm.validate_grounded_result(raw, [table_page])
|
| 154 |
+
|
| 155 |
+
assert result.valid
|
| 156 |
+
assert result.is_answered
|
| 157 |
+
assert len(result.claims) == 2
|
| 158 |
+
assert "included from day one" in result.answer.lower()
|
| 159 |
+
assert "25% off, always" in result.answer
|
| 160 |
+
assert "only" not in result.answer.lower()
|
| 161 |
+
assert result.cited_chunk_ids == ("mentorship-curriculum-table",)
|
| 162 |
+
|
| 163 |
+
|
| 164 |
+
def test_invalid_claim_text_fails_closed_even_with_valid_table_row_quotes() -> None:
|
| 165 |
+
included_quote = "10-Hour LLM Fundamentals video course $199 Included from day one"
|
| 166 |
+
discount_quote = (
|
| 167 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 168 |
+
"$349–499 each 25% off, always"
|
| 169 |
+
)
|
| 170 |
+
table_page = _with_spans(
|
| 171 |
+
{
|
| 172 |
+
"chunk_id": "mentorship-table",
|
| 173 |
+
"title": "Towards AI Mentorship",
|
| 174 |
+
"kind": "mentorship",
|
| 175 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 176 |
+
"headings": ["The curriculum comes with the team"],
|
| 177 |
+
"text": f"{included_quote} {discount_quote}",
|
| 178 |
+
},
|
| 179 |
+
included_quote,
|
| 180 |
+
discount_quote,
|
| 181 |
+
)
|
| 182 |
+
raw = json.dumps(
|
| 183 |
+
{
|
| 184 |
+
"status": "answered",
|
| 185 |
+
"claims": [
|
| 186 |
+
{
|
| 187 |
+
"text": (
|
| 188 |
+
"The mentorship includes the LLM Fundamentals course, "
|
| 189 |
+
"not two courses of your choice."
|
| 190 |
+
),
|
| 191 |
+
"chunk_id": "mentorship-table",
|
| 192 |
+
"quote": included_quote,
|
| 193 |
+
},
|
| 194 |
+
{
|
| 195 |
+
"text": (
|
| 196 |
+
"Full Stack AI Engineering, Agent Engineering, and Master "
|
| 197 |
+
"AI for Work are not included; they are discounted."
|
| 198 |
+
),
|
| 199 |
+
"chunk_id": "mentorship-table",
|
| 200 |
+
"quote": discount_quote,
|
| 201 |
+
},
|
| 202 |
+
],
|
| 203 |
+
}
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
result = llm.validate_grounded_result(raw, [table_page])
|
| 207 |
+
|
| 208 |
+
assert not result.valid
|
| 209 |
+
assert result.status == "validation_failure"
|
| 210 |
+
assert result.answer == ""
|
| 211 |
+
assert "unsupported number words" in result.validation_error
|
| 212 |
+
|
| 213 |
+
|
| 214 |
+
def test_claim_cannot_add_framing_to_an_exact_span() -> None:
|
| 215 |
+
quote = "10-Hour LLM Fundamentals video course $199 Included from day one"
|
| 216 |
+
table_page = _with_spans(
|
| 217 |
+
{
|
| 218 |
+
"chunk_id": "mentorship-table",
|
| 219 |
+
"title": "Towards AI Mentorship",
|
| 220 |
+
"kind": "mentorship",
|
| 221 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 222 |
+
"headings": ["The curriculum comes with the team"],
|
| 223 |
+
"text": quote,
|
| 224 |
+
},
|
| 225 |
+
quote,
|
| 226 |
+
)
|
| 227 |
+
raw = _model_json(
|
| 228 |
+
text=(
|
| 229 |
+
"The mentorship includes the 10-Hour LLM Fundamentals video course "
|
| 230 |
+
"from day one."
|
| 231 |
+
),
|
| 232 |
+
quote=quote,
|
| 233 |
+
chunk_id="mentorship-table",
|
| 234 |
+
)
|
| 235 |
+
|
| 236 |
+
result = llm.validate_grounded_result(raw, [table_page])
|
| 237 |
+
|
| 238 |
+
assert not result.valid
|
| 239 |
+
assert result.status == "validation_failure"
|
| 240 |
+
assert result.answer == ""
|
| 241 |
+
assert "verbatim" in result.validation_error
|
| 242 |
+
|
| 243 |
+
|
| 244 |
+
def test_claim_cannot_borrow_framing_from_an_unrelated_chunk() -> None:
|
| 245 |
+
quote = "10-Hour LLM Fundamentals video course $199 Included from day one"
|
| 246 |
+
table_page = _with_spans(
|
| 247 |
+
{
|
| 248 |
+
"chunk_id": "mentorship-table",
|
| 249 |
+
"title": "Towards AI Mentorship",
|
| 250 |
+
"kind": "mentorship",
|
| 251 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 252 |
+
"headings": ["The curriculum comes with the team"],
|
| 253 |
+
"text": quote,
|
| 254 |
+
},
|
| 255 |
+
quote,
|
| 256 |
+
)
|
| 257 |
+
unrelated_page = {
|
| 258 |
+
"chunk_id": "unrelated-premium-page",
|
| 259 |
+
"title": "Premium Executive Subscription",
|
| 260 |
+
"kind": "course",
|
| 261 |
+
"url": "https://towardsai.com/academy/unrelated/",
|
| 262 |
+
"headings": ["Guaranteed career accelerator"],
|
| 263 |
+
"text": "Unrelated material.",
|
| 264 |
+
}
|
| 265 |
+
raw = _model_json(
|
| 266 |
+
text=(
|
| 267 |
+
"The premium executive subscription includes the 10-Hour LLM "
|
| 268 |
+
"Fundamentals video course from day one."
|
| 269 |
+
),
|
| 270 |
+
quote=quote,
|
| 271 |
+
chunk_id="mentorship-table",
|
| 272 |
+
)
|
| 273 |
+
|
| 274 |
+
result = llm.validate_grounded_result(raw, [table_page, unrelated_page])
|
| 275 |
+
|
| 276 |
+
assert not result.valid
|
| 277 |
+
assert result.status == "validation_failure"
|
| 278 |
+
assert result.answer == ""
|
| 279 |
+
assert "verbatim" in result.validation_error
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
def test_not_found_is_valid_but_has_no_user_answer_or_citations() -> None:
|
| 283 |
+
result = llm.validate_grounded_result(
|
| 284 |
+
'{"status":"not_found","claims":[]}', [MENTORSHIP_PAGE]
|
| 285 |
+
)
|
| 286 |
+
|
| 287 |
+
assert result.valid
|
| 288 |
+
assert not result.is_answered
|
| 289 |
+
assert result.status == "not_found"
|
| 290 |
+
assert result.answer == ""
|
| 291 |
+
assert result.cited_chunks == ()
|
| 292 |
+
|
| 293 |
+
|
| 294 |
+
@pytest.mark.parametrize(
|
| 295 |
+
"raw",
|
| 296 |
+
[
|
| 297 |
+
"The mentorship includes two courses.",
|
| 298 |
+
'```json\n{"status":"not_found","claims":[]}\n```',
|
| 299 |
+
'{"status":"answered","claims":[}',
|
| 300 |
+
'{"status":"not_found","claims":[],"answer":"invented"}',
|
| 301 |
+
'{"status":"answered","status":"not_found","claims":[]}',
|
| 302 |
+
'{"status":"answered","claims":[],"confidence":1}',
|
| 303 |
+
],
|
| 304 |
+
)
|
| 305 |
+
def test_malformed_or_non_schema_output_fails_closed(raw: str) -> None:
|
| 306 |
+
result = llm.validate_grounded_result(raw, [MENTORSHIP_PAGE])
|
| 307 |
+
|
| 308 |
+
assert not result.valid
|
| 309 |
+
assert result.status == "validation_failure"
|
| 310 |
+
assert result.answer == ""
|
| 311 |
+
assert result.cited_chunks == ()
|
| 312 |
+
assert result.validation_error
|
| 313 |
+
|
| 314 |
+
|
| 315 |
+
def test_invented_quote_fails_closed() -> None:
|
| 316 |
+
result = llm.validate_grounded_result(
|
| 317 |
+
_model_json(
|
| 318 |
+
text="The Mentorship program includes two courses.",
|
| 319 |
+
quote="The Mentorship program includes two courses.",
|
| 320 |
+
),
|
| 321 |
+
[MENTORSHIP_PAGE],
|
| 322 |
+
)
|
| 323 |
+
|
| 324 |
+
assert result.status == "validation_failure"
|
| 325 |
+
assert result.answer == ""
|
| 326 |
+
assert "evidence span" in result.validation_error
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
def test_bad_chunk_id_fails_closed_even_when_quote_exists_elsewhere() -> None:
|
| 330 |
+
sentence = (
|
| 331 |
+
"The Mentorship program includes one course: the 10-Hour LLM "
|
| 332 |
+
"Fundamentals course."
|
| 333 |
+
)
|
| 334 |
+
result = llm.validate_grounded_result(
|
| 335 |
+
_model_json(text=sentence, quote=sentence, chunk_id="chunk_not_retrieved"),
|
| 336 |
+
[MENTORSHIP_PAGE],
|
| 337 |
+
)
|
| 338 |
+
|
| 339 |
+
assert result.status == "validation_failure"
|
| 340 |
+
assert result.answer == ""
|
| 341 |
+
assert "unknown chunk_id" in result.validation_error
|
| 342 |
+
|
| 343 |
+
|
| 344 |
+
@pytest.mark.parametrize(
|
| 345 |
+
("claim", "quote", "error_fragment"),
|
| 346 |
+
[
|
| 347 |
+
(
|
| 348 |
+
"The Mentorship program includes 2 courses.",
|
| 349 |
+
"The Mentorship program includes 1 course",
|
| 350 |
+
"numbers",
|
| 351 |
+
),
|
| 352 |
+
(
|
| 353 |
+
"The Mentorship program includes two courses.",
|
| 354 |
+
"The Mentorship program includes one course",
|
| 355 |
+
"number words",
|
| 356 |
+
),
|
| 357 |
+
(
|
| 358 |
+
"Members receive 3 live sessions every week.",
|
| 359 |
+
"Members also receive two live sessions every week.",
|
| 360 |
+
"numbers",
|
| 361 |
+
),
|
| 362 |
+
(
|
| 363 |
+
"The yearly plan saves 25%.",
|
| 364 |
+
"The yearly plan saves 24%",
|
| 365 |
+
"percentages",
|
| 366 |
+
),
|
| 367 |
+
(
|
| 368 |
+
"The plan costs $199.",
|
| 369 |
+
"The plan costs $99",
|
| 370 |
+
"currency amounts",
|
| 371 |
+
),
|
| 372 |
+
(
|
| 373 |
+
"Details are at https://towardsai.com/invented.",
|
| 374 |
+
"Details are at https://towardsai.com/academy/mentorship/",
|
| 375 |
+
"URLs",
|
| 376 |
+
),
|
| 377 |
+
],
|
| 378 |
+
)
|
| 379 |
+
def test_unsupported_critical_facts_fail_closed(
|
| 380 |
+
claim: str, quote: str, error_fragment: str
|
| 381 |
+
) -> None:
|
| 382 |
+
page = {**MENTORSHIP_PAGE, "text": f"{MENTORSHIP_PAGE['text']} {quote}"}
|
| 383 |
+
result = llm.validate_grounded_result(
|
| 384 |
+
_model_json(
|
| 385 |
+
text=claim,
|
| 386 |
+
quote=quote,
|
| 387 |
+
chunk_id=llm.chunk_id_for_page(page),
|
| 388 |
+
),
|
| 389 |
+
[page],
|
| 390 |
+
)
|
| 391 |
+
|
| 392 |
+
assert not result.valid
|
| 393 |
+
assert result.status == "validation_failure"
|
| 394 |
+
assert result.answer == ""
|
| 395 |
+
assert error_fragment in result.validation_error
|
| 396 |
+
|
| 397 |
+
|
| 398 |
+
def test_loose_paraphrase_fails_closed() -> None:
|
| 399 |
+
quote = "Members also receive two live sessions every week."
|
| 400 |
+
claim = "Experts provide unlimited private coaching and personal hiring referrals."
|
| 401 |
+
result = llm.validate_grounded_result(
|
| 402 |
+
_model_json(text=claim, quote=quote), [MENTORSHIP_PAGE]
|
| 403 |
+
)
|
| 404 |
+
|
| 405 |
+
assert not result.valid
|
| 406 |
+
assert result.status == "validation_failure"
|
| 407 |
+
assert result.answer == ""
|
| 408 |
+
assert "verbatim" in result.validation_error
|
| 409 |
+
|
| 410 |
+
|
| 411 |
+
def test_changed_negation_fails_closed() -> None:
|
| 412 |
+
quote = "The Mentorship program does not include two courses."
|
| 413 |
+
claim = "The Mentorship program does include two courses."
|
| 414 |
+
page = {**MENTORSHIP_PAGE, "text": quote}
|
| 415 |
+
result = llm.validate_grounded_result(
|
| 416 |
+
_model_json(
|
| 417 |
+
text=claim,
|
| 418 |
+
quote=quote,
|
| 419 |
+
chunk_id=llm.chunk_id_for_page(page),
|
| 420 |
+
),
|
| 421 |
+
[page],
|
| 422 |
+
)
|
| 423 |
+
|
| 424 |
+
assert not result.valid
|
| 425 |
+
assert result.status == "validation_failure"
|
| 426 |
+
assert result.answer == ""
|
| 427 |
+
assert "negation" in result.validation_error
|
| 428 |
+
|
| 429 |
+
|
| 430 |
+
def test_oversized_exact_quote_fails_closed() -> None:
|
| 431 |
+
quote = "x" * (llm.MAX_EVIDENCE_QUOTE_CHARS + 1)
|
| 432 |
+
page = {**MENTORSHIP_PAGE, "text": quote}
|
| 433 |
+
result = llm.validate_grounded_result(
|
| 434 |
+
_model_json(
|
| 435 |
+
text="Unsupported framing.",
|
| 436 |
+
quote=quote,
|
| 437 |
+
chunk_id=llm.chunk_id_for_page(page),
|
| 438 |
+
),
|
| 439 |
+
[page],
|
| 440 |
+
)
|
| 441 |
+
|
| 442 |
+
assert not result.valid
|
| 443 |
+
assert result.status == "validation_failure"
|
| 444 |
+
assert result.answer == ""
|
| 445 |
+
assert "exceeds" in result.validation_error
|
| 446 |
+
|
| 447 |
+
|
| 448 |
+
@pytest.mark.parametrize(
|
| 449 |
+
"claim",
|
| 450 |
+
[
|
| 451 |
+
"Two courses are included.",
|
| 452 |
+
"Full Stack AI Engineering is included from day one.",
|
| 453 |
+
"Agent Engineering is included from day one.",
|
| 454 |
+
],
|
| 455 |
+
)
|
| 456 |
+
def test_claim_cannot_recombine_words_from_a_table_chunk(claim: str) -> None:
|
| 457 |
+
full_table = (
|
| 458 |
+
"Two live Q&A calls every week Course Full price As a member "
|
| 459 |
+
"10-Hour LLM Fundamentals video course $199 Included from day one "
|
| 460 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 461 |
+
"$349–499 each 25% off, always"
|
| 462 |
+
)
|
| 463 |
+
included_row = "10-Hour LLM Fundamentals video course $199 Included from day one"
|
| 464 |
+
discount_row = (
|
| 465 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for Work "
|
| 466 |
+
"$349–499 each 25% off, always"
|
| 467 |
+
)
|
| 468 |
+
page = _with_spans(
|
| 469 |
+
{**MENTORSHIP_PAGE, "chunk_id": "mentorship-table", "text": full_table},
|
| 470 |
+
included_row,
|
| 471 |
+
discount_row,
|
| 472 |
+
)
|
| 473 |
+
quote = claim.removesuffix(".")
|
| 474 |
+
result = llm.validate_grounded_result(
|
| 475 |
+
_model_json(text=claim, quote=quote, chunk_id="mentorship-table"),
|
| 476 |
+
[page],
|
| 477 |
+
)
|
| 478 |
+
|
| 479 |
+
assert result.status == "validation_failure"
|
| 480 |
+
assert result.answer == ""
|
| 481 |
+
assert "evidence span" in result.validation_error
|
| 482 |
+
|
| 483 |
+
|
| 484 |
+
def test_evidence_quote_length_is_bounded() -> None:
|
| 485 |
+
quote = "A" * (llm.MAX_EVIDENCE_QUOTE_CHARS + 1)
|
| 486 |
+
page = {**MENTORSHIP_PAGE, "chunk_id": "oversized", "text": quote}
|
| 487 |
+
result = llm.validate_grounded_result(
|
| 488 |
+
_model_json(text=f"{quote}.", quote=quote, chunk_id="oversized"),
|
| 489 |
+
[page],
|
| 490 |
+
)
|
| 491 |
+
|
| 492 |
+
assert result.status == "validation_failure"
|
| 493 |
+
assert result.answer == ""
|
| 494 |
+
assert "exceeds" in result.validation_error
|
| 495 |
+
|
| 496 |
+
|
| 497 |
+
def test_claim_must_be_a_single_complete_sentence() -> None:
|
| 498 |
+
quote = (
|
| 499 |
+
"The Mentorship program includes one course: the 10-Hour LLM "
|
| 500 |
+
"Fundamentals course. Members also receive two live sessions every week."
|
| 501 |
+
)
|
| 502 |
+
result = llm.validate_grounded_result(
|
| 503 |
+
_model_json(text=quote, quote=quote), [MENTORSHIP_PAGE]
|
| 504 |
+
)
|
| 505 |
+
|
| 506 |
+
assert result.status == "validation_failure"
|
| 507 |
+
assert "complete server-defined evidence span" in result.validation_error
|
| 508 |
+
|
| 509 |
+
|
| 510 |
+
@pytest.mark.parametrize(
|
| 511 |
+
("source_span", "unsafe_substring"),
|
| 512 |
+
[
|
| 513 |
+
("No code required: learn by doing, not watching.", "code required"),
|
| 514 |
+
("Never guaranteed, always earned.", "guaranteed"),
|
| 515 |
+
("Earned, never promised.", "promised"),
|
| 516 |
+
(
|
| 517 |
+
"If you cancel the monthly plan, you can upgrade before renewal.",
|
| 518 |
+
"you can upgrade before renewal",
|
| 519 |
+
),
|
| 520 |
+
("Pay once $399 $349 one-time Save 12%", "$399"),
|
| 521 |
+
],
|
| 522 |
+
)
|
| 523 |
+
def test_arbitrary_substrings_cannot_omit_negation_or_conditions(
|
| 524 |
+
source_span: str, unsafe_substring: str
|
| 525 |
+
) -> None:
|
| 526 |
+
page = _with_spans(
|
| 527 |
+
{
|
| 528 |
+
**MENTORSHIP_PAGE,
|
| 529 |
+
"chunk_id": "atomic-source",
|
| 530 |
+
"text": source_span,
|
| 531 |
+
},
|
| 532 |
+
source_span,
|
| 533 |
+
)
|
| 534 |
+
result = llm.validate_grounded_result(
|
| 535 |
+
_model_json(
|
| 536 |
+
text=f"{unsafe_substring}.",
|
| 537 |
+
quote=unsafe_substring,
|
| 538 |
+
chunk_id="atomic-source",
|
| 539 |
+
),
|
| 540 |
+
[page],
|
| 541 |
+
)
|
| 542 |
+
|
| 543 |
+
assert result.status == "validation_failure"
|
| 544 |
+
assert result.answer == ""
|
| 545 |
+
assert "complete server-defined evidence span" in result.validation_error
|
| 546 |
+
|
| 547 |
+
|
| 548 |
+
def test_generate_grounded_answer_never_returns_raw_unvalidated_text(
|
| 549 |
+
monkeypatch: pytest.MonkeyPatch,
|
| 550 |
+
) -> None:
|
| 551 |
+
monkeypatch.setattr(
|
| 552 |
+
llm,
|
| 553 |
+
"generate_answer",
|
| 554 |
+
lambda _prompt: llm.LLMResult("An unsupported provider sentence."),
|
| 555 |
+
)
|
| 556 |
+
|
| 557 |
+
result = llm.generate_grounded_answer("prompt", [MENTORSHIP_PAGE])
|
| 558 |
+
|
| 559 |
+
assert result.status == "validation_failure"
|
| 560 |
+
assert result.answer == ""
|
| 561 |
+
|
| 562 |
+
|
| 563 |
+
def _query_bound_result(page: dict, query: str, target_offer_id: str):
|
| 564 |
+
span = page["evidence_spans"][0]["text"]
|
| 565 |
+
raw = _model_json(
|
| 566 |
+
text=span if span.endswith((".", "!", "?")) else f"{span}.",
|
| 567 |
+
quote=span,
|
| 568 |
+
chunk_id=page["chunk_id"],
|
| 569 |
+
)
|
| 570 |
+
return llm.validate_grounded_result(
|
| 571 |
+
raw,
|
| 572 |
+
[page],
|
| 573 |
+
query=query,
|
| 574 |
+
target_offer_ids=frozenset({target_offer_id}),
|
| 575 |
+
)
|
| 576 |
+
|
| 577 |
+
|
| 578 |
+
def test_query_binding_rejects_an_exact_span_from_the_wrong_offer() -> None:
|
| 579 |
+
page = _with_spans(
|
| 580 |
+
{
|
| 581 |
+
"chunk_id": "book-community",
|
| 582 |
+
"title": "Building LLMs for Production",
|
| 583 |
+
"kind": "book",
|
| 584 |
+
"offer_id": "building-llms-for-production",
|
| 585 |
+
"entity_id": "offer:building-llms-for-production",
|
| 586 |
+
"url": "https://towardsai.com/academy/building-llms-for-production/",
|
| 587 |
+
"headings": ["Community"],
|
| 588 |
+
"text": "Community access and our own AI Tutor",
|
| 589 |
+
},
|
| 590 |
+
"Community access and our own AI Tutor",
|
| 591 |
+
)
|
| 592 |
+
|
| 593 |
+
result = _query_bound_result(
|
| 594 |
+
page,
|
| 595 |
+
"Does LLM Fundamentals include community support and an AI tutor?",
|
| 596 |
+
"llm-primer",
|
| 597 |
+
)
|
| 598 |
+
|
| 599 |
+
assert result.status == "validation_failure"
|
| 600 |
+
assert "outside the requested offer boundary" in result.validation_error
|
| 601 |
+
|
| 602 |
+
|
| 603 |
+
@pytest.mark.parametrize(
|
| 604 |
+
("offer_id", "query", "span"),
|
| 605 |
+
[
|
| 606 |
+
(
|
| 607 |
+
"full-stack-ai-engineering-free-preview",
|
| 608 |
+
"Does the Full Stack free preview give me a certificate?",
|
| 609 |
+
"03 A certification that unlocks six-figure roles",
|
| 610 |
+
),
|
| 611 |
+
(
|
| 612 |
+
"mentorship",
|
| 613 |
+
"Does the monthly mentorship plan have a 30-day money-back guarantee?",
|
| 614 |
+
"Join the Mentorship 30-day money-back guarantee",
|
| 615 |
+
),
|
| 616 |
+
(
|
| 617 |
+
"get-it-all",
|
| 618 |
+
"What is the Get It All bundle price?",
|
| 619 |
+
"$1,625 combined price bought separately",
|
| 620 |
+
),
|
| 621 |
+
(
|
| 622 |
+
"agent-engineering",
|
| 623 |
+
"Does Agent Engineering include lifetime access?",
|
| 624 |
+
"Get instant access today",
|
| 625 |
+
),
|
| 626 |
+
(
|
| 627 |
+
"mentorship",
|
| 628 |
+
"What is the monthly mentorship price?",
|
| 629 |
+
"From $75/month",
|
| 630 |
+
),
|
| 631 |
+
(
|
| 632 |
+
"mentorship",
|
| 633 |
+
"What is the month-to-month mentorship price?",
|
| 634 |
+
"From $75/month",
|
| 635 |
+
),
|
| 636 |
+
(
|
| 637 |
+
"mentorship",
|
| 638 |
+
"What does mentorship cost each month?",
|
| 639 |
+
"How is this different from a $150/month mentor?",
|
| 640 |
+
),
|
| 641 |
+
(
|
| 642 |
+
"mentorship",
|
| 643 |
+
"What is the monthly mentorship price?",
|
| 644 |
+
"1× Guest workshop monthly, recorded",
|
| 645 |
+
),
|
| 646 |
+
(
|
| 647 |
+
"mentorship",
|
| 648 |
+
"What is the yearly mentorship price?",
|
| 649 |
+
"Yearly Save 24% · $289 off",
|
| 650 |
+
),
|
| 651 |
+
(
|
| 652 |
+
"mentorship",
|
| 653 |
+
"Is the yearly mentorship plan 24% off?",
|
| 654 |
+
(
|
| 655 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for "
|
| 656 |
+
"Work $349–499 each 25% off, always"
|
| 657 |
+
),
|
| 658 |
+
),
|
| 659 |
+
(
|
| 660 |
+
"mentorship",
|
| 661 |
+
"Does monthly mentorship have a discount?",
|
| 662 |
+
(
|
| 663 |
+
"Full Stack AI Engineering · Agent Engineering · Master AI for "
|
| 664 |
+
"Work $349–499 each 25% off, always"
|
| 665 |
+
),
|
| 666 |
+
),
|
| 667 |
+
],
|
| 668 |
+
)
|
| 669 |
+
def test_query_binding_rejects_wrong_qualifier_within_the_right_offer(
|
| 670 |
+
offer_id: str, query: str, span: str
|
| 671 |
+
) -> None:
|
| 672 |
+
page = _with_spans(
|
| 673 |
+
{
|
| 674 |
+
"chunk_id": f"{offer_id}-unsafe-field",
|
| 675 |
+
"title": offer_id,
|
| 676 |
+
"kind": "course",
|
| 677 |
+
"offer_id": offer_id,
|
| 678 |
+
"entity_id": f"offer:{offer_id}",
|
| 679 |
+
"url": f"https://towardsai.com/academy/{offer_id}/",
|
| 680 |
+
"headings": ["Offer"],
|
| 681 |
+
"text": span,
|
| 682 |
+
},
|
| 683 |
+
span,
|
| 684 |
+
)
|
| 685 |
+
|
| 686 |
+
result = _query_bound_result(page, query, offer_id)
|
| 687 |
+
|
| 688 |
+
assert result.status == "validation_failure"
|
| 689 |
+
assert "target-qualified" in result.validation_error
|
| 690 |
+
|
| 691 |
+
|
| 692 |
+
@pytest.mark.parametrize(
|
| 693 |
+
("offer_id", "query", "span", "targets"),
|
| 694 |
+
[
|
| 695 |
+
(
|
| 696 |
+
"full-stack-ai-engineering",
|
| 697 |
+
"How many Full Stack AI Engineering lessons can I preview for free?",
|
| 698 |
+
"Explore the first 6 lessons free.",
|
| 699 |
+
frozenset(
|
| 700 |
+
{
|
| 701 |
+
"full-stack-ai-engineering",
|
| 702 |
+
"full-stack-ai-engineering-free-preview",
|
| 703 |
+
}
|
| 704 |
+
),
|
| 705 |
+
),
|
| 706 |
+
(
|
| 707 |
+
"llm-primer",
|
| 708 |
+
"Is LLM Fundamentals 10 or 12 hours long?",
|
| 709 |
+
(
|
| 710 |
+
"It is self-paced; the average completion time is 10 hours "
|
| 711 |
+
"across five 2-hour sessions."
|
| 712 |
+
),
|
| 713 |
+
frozenset({"llm-primer"}),
|
| 714 |
+
),
|
| 715 |
+
(
|
| 716 |
+
"mentorship",
|
| 717 |
+
"What is the month-to-month mentorship price?",
|
| 718 |
+
"$99 /month",
|
| 719 |
+
frozenset({"mentorship"}),
|
| 720 |
+
),
|
| 721 |
+
],
|
| 722 |
+
)
|
| 723 |
+
def test_query_binding_accepts_exact_target_qualified_facts(
|
| 724 |
+
offer_id: str,
|
| 725 |
+
query: str,
|
| 726 |
+
span: str,
|
| 727 |
+
targets: frozenset[str],
|
| 728 |
+
) -> None:
|
| 729 |
+
page = _with_spans(
|
| 730 |
+
{
|
| 731 |
+
"chunk_id": f"{offer_id}-valid-field",
|
| 732 |
+
"title": offer_id,
|
| 733 |
+
"kind": "course",
|
| 734 |
+
"offer_id": offer_id,
|
| 735 |
+
"entity_id": f"offer:{offer_id}",
|
| 736 |
+
"url": f"https://towardsai.com/academy/{offer_id}/",
|
| 737 |
+
"headings": ["Offer"],
|
| 738 |
+
"text": span,
|
| 739 |
+
},
|
| 740 |
+
span,
|
| 741 |
+
)
|
| 742 |
+
raw = _model_json(
|
| 743 |
+
text=span,
|
| 744 |
+
quote=span,
|
| 745 |
+
chunk_id=page["chunk_id"],
|
| 746 |
+
)
|
| 747 |
+
|
| 748 |
+
result = llm.validate_grounded_result(
|
| 749 |
+
raw,
|
| 750 |
+
[page],
|
| 751 |
+
query=query,
|
| 752 |
+
target_offer_ids=targets,
|
| 753 |
+
)
|
| 754 |
+
|
| 755 |
+
assert result.is_answered
|
| 756 |
+
assert result.answer == (
|
| 757 |
+
span if span.endswith((".", "!", "?")) else f"{span}."
|
| 758 |
+
)
|
| 759 |
+
|
| 760 |
+
|
| 761 |
+
def test_preview_count_exception_rejects_paid_course_access_claim() -> None:
|
| 762 |
+
preview_count = "Free preview · 7 full lessons"
|
| 763 |
+
paid_access = "Unlock Lifetime Access"
|
| 764 |
+
preview_page = _with_spans(
|
| 765 |
+
{
|
| 766 |
+
"chunk_id": "agent-preview-count",
|
| 767 |
+
"title": "Agent Engineering free preview",
|
| 768 |
+
"kind": "course_preview",
|
| 769 |
+
"offer_id": "agent-engineering-free-preview",
|
| 770 |
+
"entity_id": "offer:agent-engineering-free-preview",
|
| 771 |
+
"url": (
|
| 772 |
+
"https://towardsai.com/academy/"
|
| 773 |
+
"agent-engineering-free-preview/"
|
| 774 |
+
),
|
| 775 |
+
"headings": ["Free preview"],
|
| 776 |
+
"text": preview_count,
|
| 777 |
+
},
|
| 778 |
+
preview_count,
|
| 779 |
+
)
|
| 780 |
+
paid_page = _with_spans(
|
| 781 |
+
{
|
| 782 |
+
"chunk_id": "agent-paid-access",
|
| 783 |
+
"title": "Agent Engineering",
|
| 784 |
+
"kind": "course",
|
| 785 |
+
"offer_id": "agent-engineering",
|
| 786 |
+
"entity_id": "offer:agent-engineering",
|
| 787 |
+
"url": "https://towardsai.com/academy/agent-engineering/",
|
| 788 |
+
"headings": ["Purchase"],
|
| 789 |
+
"text": paid_access,
|
| 790 |
+
},
|
| 791 |
+
paid_access,
|
| 792 |
+
)
|
| 793 |
+
raw = json.dumps(
|
| 794 |
+
{
|
| 795 |
+
"status": "answered",
|
| 796 |
+
"claims": [
|
| 797 |
+
{
|
| 798 |
+
"text": preview_count,
|
| 799 |
+
"chunk_id": preview_page["chunk_id"],
|
| 800 |
+
"quote": preview_count,
|
| 801 |
+
},
|
| 802 |
+
{
|
| 803 |
+
"text": paid_access,
|
| 804 |
+
"chunk_id": paid_page["chunk_id"],
|
| 805 |
+
"quote": paid_access,
|
| 806 |
+
},
|
| 807 |
+
],
|
| 808 |
+
}
|
| 809 |
+
)
|
| 810 |
+
query = (
|
| 811 |
+
"Do the 7 lessons in the Agent Engineering preview include lifetime access?"
|
| 812 |
+
)
|
| 813 |
+
|
| 814 |
+
result = llm.validate_grounded_result(
|
| 815 |
+
raw,
|
| 816 |
+
[preview_page, paid_page],
|
| 817 |
+
query=query,
|
| 818 |
+
target_offer_ids=frozenset(
|
| 819 |
+
{"agent-engineering-free-preview", "agent-engineering"}
|
| 820 |
+
),
|
| 821 |
+
)
|
| 822 |
+
|
| 823 |
+
assert result.status == "validation_failure"
|
| 824 |
+
assert "target-qualified" in result.validation_error
|
| 825 |
+
|
| 826 |
+
|
| 827 |
+
@pytest.mark.parametrize(
|
| 828 |
+
("query", "span"),
|
| 829 |
+
[
|
| 830 |
+
(
|
| 831 |
+
"Does mentorship include lifetime course access?",
|
| 832 |
+
(
|
| 833 |
+
"A senior AI consultant runs $200–500 an hour, with retainers "
|
| 834 |
+
"from $3,000 a month."
|
| 835 |
+
),
|
| 836 |
+
),
|
| 837 |
+
(
|
| 838 |
+
"Does mentorship include resume and project reviews?",
|
| 839 |
+
(
|
| 840 |
+
"A single mentor runs $120–450 a month, with resume reviews "
|
| 841 |
+
"billed $150–300 each."
|
| 842 |
+
),
|
| 843 |
+
),
|
| 844 |
+
],
|
| 845 |
+
)
|
| 846 |
+
def test_competitor_comparison_spans_are_never_offer_evidence(
|
| 847 |
+
query: str, span: str
|
| 848 |
+
) -> None:
|
| 849 |
+
page = _with_spans(
|
| 850 |
+
{
|
| 851 |
+
"chunk_id": "mentorship-comparison",
|
| 852 |
+
"title": "Towards AI Mentorship",
|
| 853 |
+
"kind": "mentorship",
|
| 854 |
+
"offer_id": "mentorship",
|
| 855 |
+
"entity_id": "offer:mentorship",
|
| 856 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 857 |
+
"headings": ["Comparison"],
|
| 858 |
+
"text": span,
|
| 859 |
+
},
|
| 860 |
+
span,
|
| 861 |
+
)
|
| 862 |
+
|
| 863 |
+
result = _query_bound_result(page, query, "mentorship")
|
| 864 |
+
|
| 865 |
+
assert result.status == "validation_failure"
|
| 866 |
+
assert "competitor comparison" in result.validation_error
|
tests/test_live_smoke.py
CHANGED
|
@@ -7,13 +7,16 @@ import pytest
|
|
| 7 |
import requests
|
| 8 |
|
| 9 |
LIVE_BASE_URL = os.getenv("LIVE_SPACE_BASE_URL", "").rstrip("/")
|
|
|
|
|
|
|
|
|
|
| 10 |
RUN_LIVE_CHAT = os.getenv("RUN_LIVE_CHAT_SMOKE", "").lower() in {
|
| 11 |
"1",
|
| 12 |
"true",
|
| 13 |
"yes",
|
| 14 |
"on",
|
| 15 |
}
|
| 16 |
-
HEADERS = {"Origin": "https://
|
| 17 |
FIRST_PROMPT = "I want help deciding which course to take."
|
| 18 |
|
| 19 |
|
|
@@ -42,7 +45,9 @@ def test_live_health_widget_and_config() -> None:
|
|
| 42 |
base_url = require_live_base_url()
|
| 43 |
|
| 44 |
health, health_seconds = timed_request("GET", f"{base_url}/healthz", timeout=120)
|
| 45 |
-
widget, widget_seconds = timed_request(
|
|
|
|
|
|
|
| 46 |
config, config_seconds = timed_request(
|
| 47 |
"GET", f"{base_url}/api/helper/config", timeout=120
|
| 48 |
)
|
|
@@ -58,7 +63,14 @@ def test_live_health_widget_and_config() -> None:
|
|
| 58 |
assert config.status_code == 200
|
| 59 |
payload = config.json()
|
| 60 |
assert FIRST_PROMPT in payload["forcedPrompts"]
|
| 61 |
-
assert "
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
assert config_seconds <= max_seconds("LIVE_SMOKE_MAX_CONFIG_SECONDS", 30)
|
| 63 |
|
| 64 |
|
|
@@ -77,7 +89,7 @@ def test_live_chat_returns_concise_answer() -> None:
|
|
| 77 |
"selectedPrompt": FIRST_PROMPT,
|
| 78 |
"visitorId": "github-action-smoke",
|
| 79 |
"context": {
|
| 80 |
-
"url": "https://
|
| 81 |
"pageTitle": "Agent course",
|
| 82 |
"signedIn": False,
|
| 83 |
},
|
|
@@ -92,3 +104,53 @@ def test_live_chat_returns_concise_answer() -> None:
|
|
| 92 |
assert len(payload["answer"]) < 1400
|
| 93 |
assert payload["sources"]
|
| 94 |
assert elapsed <= max_seconds("LIVE_SMOKE_MAX_CHAT_SECONDS", 90)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
import requests
|
| 8 |
|
| 9 |
LIVE_BASE_URL = os.getenv("LIVE_SPACE_BASE_URL", "").rstrip("/")
|
| 10 |
+
LIVE_HELPER_WIDGET_PATH = "/" + os.getenv(
|
| 11 |
+
"LIVE_HELPER_WIDGET_PATH", "/helper-widget.js"
|
| 12 |
+
).strip("/")
|
| 13 |
RUN_LIVE_CHAT = os.getenv("RUN_LIVE_CHAT_SMOKE", "").lower() in {
|
| 14 |
"1",
|
| 15 |
"true",
|
| 16 |
"yes",
|
| 17 |
"on",
|
| 18 |
}
|
| 19 |
+
HEADERS = {"Origin": "https://towardsai.com"}
|
| 20 |
FIRST_PROMPT = "I want help deciding which course to take."
|
| 21 |
|
| 22 |
|
|
|
|
| 45 |
base_url = require_live_base_url()
|
| 46 |
|
| 47 |
health, health_seconds = timed_request("GET", f"{base_url}/healthz", timeout=120)
|
| 48 |
+
widget, widget_seconds = timed_request(
|
| 49 |
+
"GET", f"{base_url}{LIVE_HELPER_WIDGET_PATH}", timeout=120
|
| 50 |
+
)
|
| 51 |
config, config_seconds = timed_request(
|
| 52 |
"GET", f"{base_url}/api/helper/config", timeout=120
|
| 53 |
)
|
|
|
|
| 63 |
assert config.status_code == 200
|
| 64 |
payload = config.json()
|
| 65 |
assert FIRST_PROMPT in payload["forcedPrompts"]
|
| 66 |
+
assert "towardsai.com" in payload["siteWideHosts"]
|
| 67 |
+
assert (
|
| 68 |
+
"/academy/agent-engineering" in payload["allowedPathsByHost"]["towardsai.com"]
|
| 69 |
+
)
|
| 70 |
+
assert (
|
| 71 |
+
"/courses/agent-engineering"
|
| 72 |
+
in payload["allowedPathsByHost"]["academy.towardsai.net"]
|
| 73 |
+
)
|
| 74 |
assert config_seconds <= max_seconds("LIVE_SMOKE_MAX_CONFIG_SECONDS", 30)
|
| 75 |
|
| 76 |
|
|
|
|
| 89 |
"selectedPrompt": FIRST_PROMPT,
|
| 90 |
"visitorId": "github-action-smoke",
|
| 91 |
"context": {
|
| 92 |
+
"url": "https://towardsai.com/academy/agent-engineering/",
|
| 93 |
"pageTitle": "Agent course",
|
| 94 |
"signedIn": False,
|
| 95 |
},
|
|
|
|
| 104 |
assert len(payload["answer"]) < 1400
|
| 105 |
assert payload["sources"]
|
| 106 |
assert elapsed <= max_seconds("LIVE_SMOKE_MAX_CHAT_SECONDS", 90)
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
@pytest.mark.live
|
| 110 |
+
@pytest.mark.skipif(
|
| 111 |
+
not RUN_LIVE_CHAT,
|
| 112 |
+
reason="RUN_LIVE_CHAT_SMOKE is not enabled",
|
| 113 |
+
)
|
| 114 |
+
def test_live_chat_fails_closed_for_offer_access_and_guarantees() -> None:
|
| 115 |
+
base_url = require_live_base_url()
|
| 116 |
+
|
| 117 |
+
def ask(query: str, visitor_id: str) -> dict:
|
| 118 |
+
response = requests.post(
|
| 119 |
+
f"{base_url}/api/helper/chat",
|
| 120 |
+
json={
|
| 121 |
+
"query": query,
|
| 122 |
+
"selectedPrompt": FIRST_PROMPT,
|
| 123 |
+
"visitorId": visitor_id,
|
| 124 |
+
"threadId": "",
|
| 125 |
+
"history": [{"role": "user", "content": FIRST_PROMPT}],
|
| 126 |
+
"context": {
|
| 127 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 128 |
+
"pageTitle": "Mentorship",
|
| 129 |
+
"signedIn": False,
|
| 130 |
+
},
|
| 131 |
+
},
|
| 132 |
+
headers=HEADERS,
|
| 133 |
+
timeout=120,
|
| 134 |
+
)
|
| 135 |
+
assert response.status_code == 200
|
| 136 |
+
return response.json()
|
| 137 |
+
|
| 138 |
+
incident = ask(
|
| 139 |
+
"Does mentorship include access to 2 courses of my choice?",
|
| 140 |
+
"github-action-mentorship-grounding",
|
| 141 |
+
)
|
| 142 |
+
assert incident["status"] == "answered"
|
| 143 |
+
assert "Included from day one" in incident["answer"]
|
| 144 |
+
assert "25% off, always" in incident["answer"]
|
| 145 |
+
assert "two courses" not in incident["answer"].casefold()
|
| 146 |
+
assert {source["url"] for source in incident["sources"]} == {
|
| 147 |
+
"https://towardsai.com/academy/mentorship/"
|
| 148 |
+
}
|
| 149 |
+
|
| 150 |
+
unsupported = ask(
|
| 151 |
+
"Does Agent Engineering guarantee me a job and include a $500 discount?",
|
| 152 |
+
"github-action-unsupported-offer-fact",
|
| 153 |
+
)
|
| 154 |
+
assert unsupported["status"] == "insufficient_evidence"
|
| 155 |
+
assert unsupported["sources"] == []
|
| 156 |
+
assert "https://towardsai.com/academy/contact/#contact" in unsupported["answer"]
|
tests/test_llm.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
from typing import Any
|
| 4 |
+
|
| 5 |
+
from tai_helper import llm
|
| 6 |
+
from tai_helper.settings import Settings
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class FakeDeepSeekResponse:
|
| 10 |
+
def __init__(self, status_code: int, payload: dict[str, Any], text: str = ""):
|
| 11 |
+
self.status_code = status_code
|
| 12 |
+
self._payload = payload
|
| 13 |
+
self.text = text
|
| 14 |
+
|
| 15 |
+
def json(self) -> dict[str, Any]:
|
| 16 |
+
return self._payload
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
class FakeUsageMetadata:
|
| 20 |
+
prompt_token_count = 5
|
| 21 |
+
candidates_token_count = 7
|
| 22 |
+
total_token_count = 12
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class FakeGeminiResponse:
|
| 26 |
+
text = "Gemini fallback answer"
|
| 27 |
+
usage_metadata = FakeUsageMetadata()
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
class FakeGeminiModels:
|
| 31 |
+
def __init__(self) -> None:
|
| 32 |
+
self.calls: list[dict[str, Any]] = []
|
| 33 |
+
|
| 34 |
+
def generate_content(self, **kwargs) -> FakeGeminiResponse:
|
| 35 |
+
self.calls.append(kwargs)
|
| 36 |
+
return FakeGeminiResponse()
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
class FakeGeminiClient:
|
| 40 |
+
models = FakeGeminiModels()
|
| 41 |
+
|
| 42 |
+
def __init__(self, api_key: str) -> None:
|
| 43 |
+
self.api_key = api_key
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
def test_prompt_requires_direct_support_for_exact_offer_claims() -> None:
|
| 47 |
+
prompt = llm.build_prompt(
|
| 48 |
+
query="How many courses are included in mentorship?",
|
| 49 |
+
selected_prompt="I want to find mentors",
|
| 50 |
+
current_url="https://towardsai.com/academy/mentorship/",
|
| 51 |
+
page_title="Towards AI Mentorship",
|
| 52 |
+
history=[],
|
| 53 |
+
selected_pages=[
|
| 54 |
+
{
|
| 55 |
+
"title": "Towards AI Mentorship",
|
| 56 |
+
"kind": "mentorship",
|
| 57 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 58 |
+
"headings": ["The curriculum comes with the team"],
|
| 59 |
+
"text": (
|
| 60 |
+
"Towards AI Mentorship includes one course: the 10-Hour LLM "
|
| 61 |
+
"Fundamentals video course from day one."
|
| 62 |
+
),
|
| 63 |
+
}
|
| 64 |
+
],
|
| 65 |
+
)
|
| 66 |
+
|
| 67 |
+
instruction = llm.SYSTEM_INSTRUCTION.lower()
|
| 68 |
+
assert "only authority for factual claims" in instruction
|
| 69 |
+
assert "number or names of included products or courses" in instruction
|
| 70 |
+
assert "do not guess" in instruction
|
| 71 |
+
assert "every factual claim must be directly supported" in prompt.lower()
|
| 72 |
+
assert "cannot confirm it" in prompt.lower()
|
| 73 |
+
assert "includes one course" in prompt
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def test_prompt_corrects_unsupported_alternatives_with_supported_table_facts() -> None:
|
| 77 |
+
prompt = llm.build_prompt(
|
| 78 |
+
query=(
|
| 79 |
+
"Does the mentorship include two courses of our choice, or only the "
|
| 80 |
+
"LLM Fundamentals course?"
|
| 81 |
+
),
|
| 82 |
+
selected_prompt="I want to find mentors",
|
| 83 |
+
current_url="https://towardsai.com/academy/mentorship/",
|
| 84 |
+
page_title="Towards AI Mentorship",
|
| 85 |
+
history=[],
|
| 86 |
+
selected_pages=[
|
| 87 |
+
{
|
| 88 |
+
"chunk_id": "mentorship-curriculum-table",
|
| 89 |
+
"title": "Towards AI Mentorship",
|
| 90 |
+
"kind": "mentorship",
|
| 91 |
+
"url": "https://towardsai.com/academy/mentorship/",
|
| 92 |
+
"headings": ["The curriculum comes with the team"],
|
| 93 |
+
"text": (
|
| 94 |
+
"10-Hour LLM Fundamentals video course $199 Included from "
|
| 95 |
+
"day one Full Stack AI Engineering · Agent Engineering · "
|
| 96 |
+
"Master AI for Work $349–499 each 25% off, always"
|
| 97 |
+
),
|
| 98 |
+
}
|
| 99 |
+
],
|
| 100 |
+
)
|
| 101 |
+
|
| 102 |
+
instruction = llm.SYSTEM_INSTRUCTION.lower()
|
| 103 |
+
assert "unsupported premise or alternative" in instruction
|
| 104 |
+
assert "do not return" in instruction
|
| 105 |
+
assert "merely because the visitor's proposed premise is unsupported" in instruction
|
| 106 |
+
assert 'never infer a total, use "only"' in instruction
|
| 107 |
+
prompt_lower = prompt.lower()
|
| 108 |
+
assert "correction rule" in prompt_lower
|
| 109 |
+
assert (
|
| 110 |
+
"return answered with those facts as separate extractive claims" in prompt_lower
|
| 111 |
+
)
|
| 112 |
+
assert "one table row says an item is included" in prompt_lower
|
| 113 |
+
assert 'do not infer a total or say "only"' in prompt_lower
|
| 114 |
+
assert "do not repeat or deny the visitor's unsupported premise" in prompt_lower
|
| 115 |
+
assert 'never write "not included"' in prompt_lower
|
| 116 |
+
assert (
|
| 117 |
+
"unless those literal words occur in that claim's exact quote" in prompt_lower
|
| 118 |
+
)
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def test_generate_answer_uses_deepseek_primary(monkeypatch) -> None:
|
| 122 |
+
request_calls = []
|
| 123 |
+
monkeypatch.setattr(
|
| 124 |
+
llm,
|
| 125 |
+
"settings",
|
| 126 |
+
Settings(deepseek_api_key="deepseek-key", gemini_api_key="gemini-key"),
|
| 127 |
+
)
|
| 128 |
+
|
| 129 |
+
def fake_post(*args, **kwargs):
|
| 130 |
+
request_calls.append({"args": args, **kwargs})
|
| 131 |
+
return FakeDeepSeekResponse(
|
| 132 |
+
200,
|
| 133 |
+
{
|
| 134 |
+
"choices": [{"message": {"content": "DeepSeek answer"}}],
|
| 135 |
+
"usage": {
|
| 136 |
+
"prompt_tokens": 10,
|
| 137 |
+
"completion_tokens": 8,
|
| 138 |
+
"total_tokens": 18,
|
| 139 |
+
},
|
| 140 |
+
},
|
| 141 |
+
)
|
| 142 |
+
|
| 143 |
+
monkeypatch.setattr(llm.requests, "post", fake_post)
|
| 144 |
+
|
| 145 |
+
result = llm.generate_answer("Visitor prompt")
|
| 146 |
+
|
| 147 |
+
assert result.answer == "DeepSeek answer"
|
| 148 |
+
assert result.usage["provider"] == "deepseek"
|
| 149 |
+
assert result.usage["model"] == "deepseek-v4-flash"
|
| 150 |
+
assert result.usage["total_tokens"] == 18
|
| 151 |
+
assert request_calls[0]["args"] == ("https://api.deepseek.com/chat/completions",)
|
| 152 |
+
assert request_calls[0]["headers"] == {
|
| 153 |
+
"Authorization": "Bearer deepseek-key",
|
| 154 |
+
"Content-Type": "application/json",
|
| 155 |
+
}
|
| 156 |
+
assert request_calls[0]["json"]["model"] == "deepseek-v4-flash"
|
| 157 |
+
assert request_calls[0]["json"]["thinking"] == {"type": "disabled"}
|
| 158 |
+
assert request_calls[0]["json"]["messages"][0]["role"] == "system"
|
| 159 |
+
assert request_calls[0]["json"]["messages"][1] == {
|
| 160 |
+
"role": "user",
|
| 161 |
+
"content": "Visitor prompt",
|
| 162 |
+
}
|
| 163 |
+
|
| 164 |
+
|
| 165 |
+
def test_generate_answer_falls_back_to_gemini_when_deepseek_fails(
|
| 166 |
+
monkeypatch,
|
| 167 |
+
) -> None:
|
| 168 |
+
FakeGeminiClient.models = FakeGeminiModels()
|
| 169 |
+
monkeypatch.setattr(
|
| 170 |
+
llm,
|
| 171 |
+
"settings",
|
| 172 |
+
Settings(deepseek_api_key="deepseek-key", gemini_api_key="gemini-key"),
|
| 173 |
+
)
|
| 174 |
+
monkeypatch.setattr(
|
| 175 |
+
llm.requests,
|
| 176 |
+
"post",
|
| 177 |
+
lambda *_args, **_kwargs: FakeDeepSeekResponse(
|
| 178 |
+
503, {}, "temporarily unavailable"
|
| 179 |
+
),
|
| 180 |
+
)
|
| 181 |
+
monkeypatch.setattr(llm.genai, "Client", FakeGeminiClient)
|
| 182 |
+
|
| 183 |
+
result = llm.generate_answer("Visitor prompt")
|
| 184 |
+
|
| 185 |
+
assert result.answer == "Gemini fallback answer"
|
| 186 |
+
assert result.usage["provider"] == "google_genai"
|
| 187 |
+
assert result.usage["model"] == "gemini-2.5-flash"
|
| 188 |
+
assert result.usage["fallback_from"] == "deepseek"
|
| 189 |
+
assert FakeGeminiClient.models.calls[0]["model"] == "gemini-2.5-flash"
|
tests/test_offers.py
ADDED
|
@@ -0,0 +1,753 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import json
|
| 4 |
+
from urllib.parse import urlparse
|
| 5 |
+
|
| 6 |
+
import pytest
|
| 7 |
+
|
| 8 |
+
from tai_helper import llm
|
| 9 |
+
from tai_helper.catalog import pages, retrieve
|
| 10 |
+
from tai_helper.settings import repo_root
|
| 11 |
+
|
| 12 |
+
CURRENT_COM_URLS = {
|
| 13 |
+
"https://towardsai.com/",
|
| 14 |
+
"https://towardsai.com/academy/",
|
| 15 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 16 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 17 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 18 |
+
"https://towardsai.com/academy/python-for-ai-engineering/",
|
| 19 |
+
"https://towardsai.com/academy/ai-for-work/",
|
| 20 |
+
"https://towardsai.com/academy/building-llms-for-production/",
|
| 21 |
+
"https://towardsai.com/academy/bundles/",
|
| 22 |
+
"https://towardsai.com/academy/mentorship/",
|
| 23 |
+
"https://towardsai.com/academy/about/",
|
| 24 |
+
"https://towardsai.com/academy/contact/",
|
| 25 |
+
"https://towardsai.com/contribute/",
|
| 26 |
+
"https://towardsai.com/enterprise/software-developer-to-ai-engineer/",
|
| 27 |
+
"https://towardsai.com/enterprise/agentic-developer-conversion/",
|
| 28 |
+
"https://towardsai.com/academy/affiliate/",
|
| 29 |
+
"https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 30 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 31 |
+
"https://towardsai.com/academy/book/",
|
| 32 |
+
"https://towardsai.com/academy/bundles/get-it-all/",
|
| 33 |
+
(
|
| 34 |
+
"https://towardsai.com/academy/bundles/"
|
| 35 |
+
"from-coding-novice-to-advanced-llm-developer/"
|
| 36 |
+
),
|
| 37 |
+
(
|
| 38 |
+
"https://towardsai.com/academy/bundles/"
|
| 39 |
+
"10-hour-crash-course-into-llm-developer-expert/"
|
| 40 |
+
),
|
| 41 |
+
"https://towardsai.com/enterpriseenablement/",
|
| 42 |
+
"https://towardsai.com/valuecreation/",
|
| 43 |
+
"https://towardsai.com/valuecreation/careers/",
|
| 44 |
+
"https://towardsai.com/valuecreation/deployment-strategist/",
|
| 45 |
+
"https://towardsai.com/valuecreation/junior-ai-engineer/",
|
| 46 |
+
"https://towardsai.com/valuecreation/senior-ai-engineer/",
|
| 47 |
+
"https://towardsai.com/theaitastegap/",
|
| 48 |
+
"https://towardsai.com/webinars/agentengineering/",
|
| 49 |
+
}
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _catalog(name: str) -> dict:
|
| 53 |
+
return json.loads((repo_root() / "data" / name).read_text())
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def _page_text(url: str) -> str:
|
| 57 |
+
matching = [page for page in pages() if page["url"] == url]
|
| 58 |
+
assert len(matching) == 1, f"expected one active canonical page for {url}"
|
| 59 |
+
return matching[0]["text"].lower()
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def test_live_refresh_scans_every_current_towardsai_com_page() -> None:
|
| 63 |
+
catalog = _catalog("towardsai_com_pages.json")
|
| 64 |
+
|
| 65 |
+
assert {page["discovered_url"] for page in catalog["pages"]} == CURRENT_COM_URLS
|
| 66 |
+
assert {page["url"] for page in catalog["pages"]} == CURRENT_COM_URLS
|
| 67 |
+
assert catalog["status_counts"] == {"excluded": 3, "included": 27}
|
| 68 |
+
excluded_urls = {
|
| 69 |
+
page["url"] for page in catalog["pages"] if not page["retrieval_eligible"]
|
| 70 |
+
}
|
| 71 |
+
assert excluded_urls == {
|
| 72 |
+
"https://towardsai.com/",
|
| 73 |
+
"https://towardsai.com/academy/",
|
| 74 |
+
"https://towardsai.com/academy/bundles/",
|
| 75 |
+
}
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
def test_academy_refresh_scans_sitemap_and_excludes_unsafe_pages() -> None:
|
| 79 |
+
catalog = _catalog("pages.json")
|
| 80 |
+
academy_pages = [
|
| 81 |
+
page
|
| 82 |
+
for page in catalog["pages"]
|
| 83 |
+
if page["discovered_url"].startswith("https://academy.towardsai.net")
|
| 84 |
+
]
|
| 85 |
+
excluded_paths = {
|
| 86 |
+
urlparse(page["discovered_url"]).path.rstrip("/") or "/"
|
| 87 |
+
for page in academy_pages
|
| 88 |
+
if not page["retrieval_eligible"]
|
| 89 |
+
}
|
| 90 |
+
|
| 91 |
+
assert len(academy_pages) == 28
|
| 92 |
+
assert {
|
| 93 |
+
"/",
|
| 94 |
+
"/pages/agent-course-landing-page",
|
| 95 |
+
"/pages/towards-ai-insider",
|
| 96 |
+
"/pages/agent-course-new-page-cro",
|
| 97 |
+
"/pages/webinar",
|
| 98 |
+
"/pages/new-home-page",
|
| 99 |
+
"/pages/landing-page-free-email-course",
|
| 100 |
+
"/pages/choose-your-course",
|
| 101 |
+
"/collections",
|
| 102 |
+
"/collections/products",
|
| 103 |
+
"/collections/developers",
|
| 104 |
+
"/collections/professionals",
|
| 105 |
+
"/pages/free-resources",
|
| 106 |
+
} <= excluded_paths
|
| 107 |
+
assert not any(
|
| 108 |
+
page["retrieval_eligible"]
|
| 109 |
+
for page in catalog["pages"]
|
| 110 |
+
if page["authority"] == "curated_external"
|
| 111 |
+
)
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
def test_every_evidence_page_has_fresh_provenance_and_real_chunks() -> None:
|
| 115 |
+
for filename in ("towardsai_com_pages.json", "pages.json"):
|
| 116 |
+
for page in _catalog(filename)["pages"]:
|
| 117 |
+
if not page["retrieval_eligible"]:
|
| 118 |
+
continue
|
| 119 |
+
assert page["status"] == "included"
|
| 120 |
+
assert page["fetched_at"]
|
| 121 |
+
assert len(page["content_hash"]) == 64
|
| 122 |
+
assert page["http_status"] == 200
|
| 123 |
+
assert page["chunks"]
|
| 124 |
+
assert all(chunk["chunk_id"] and chunk["text"] for chunk in page["chunks"])
|
| 125 |
+
assert all(len(chunk["text"]) <= 1800 for chunk in page["chunks"])
|
| 126 |
+
assert all(chunk["evidence_spans"] for chunk in page["chunks"])
|
| 127 |
+
assert all(
|
| 128 |
+
span["text"] in " ".join(chunk["text"].split())
|
| 129 |
+
for chunk in page["chunks"]
|
| 130 |
+
for span in chunk["evidence_spans"]
|
| 131 |
+
)
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
@pytest.mark.parametrize(
|
| 135 |
+
("url", "facts"),
|
| 136 |
+
[
|
| 137 |
+
(
|
| 138 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 139 |
+
("$349 one-time", "92 lessons", "60+ hrs"),
|
| 140 |
+
),
|
| 141 |
+
(
|
| 142 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 143 |
+
("$499 one-time", "35 lessons", "two production agents"),
|
| 144 |
+
),
|
| 145 |
+
(
|
| 146 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 147 |
+
("$199 one-time", "5 in-depth 2-hour video sessions"),
|
| 148 |
+
),
|
| 149 |
+
(
|
| 150 |
+
"https://towardsai.com/academy/python-for-ai-engineering/",
|
| 151 |
+
("$149 one-time", "38 lessons", "complete beginners welcome"),
|
| 152 |
+
),
|
| 153 |
+
(
|
| 154 |
+
"https://towardsai.com/academy/ai-for-work/",
|
| 155 |
+
("$399 one-time", "98 lessons", "no code required"),
|
| 156 |
+
),
|
| 157 |
+
(
|
| 158 |
+
"https://towardsai.com/academy/building-llms-for-production/",
|
| 159 |
+
("$29.99", "470-page", "84 lessons"),
|
| 160 |
+
),
|
| 161 |
+
(
|
| 162 |
+
"https://towardsai.com/academy/mentorship/",
|
| 163 |
+
(
|
| 164 |
+
"10-hour llm fundamentals video course",
|
| 165 |
+
"included from day one",
|
| 166 |
+
"25% off, always",
|
| 167 |
+
),
|
| 168 |
+
),
|
| 169 |
+
(
|
| 170 |
+
"https://towardsai.com/academy/bundles/get-it-all/",
|
| 171 |
+
("6 products in the bundle", "$1,625 combined price"),
|
| 172 |
+
),
|
| 173 |
+
(
|
| 174 |
+
(
|
| 175 |
+
"https://towardsai.com/academy/bundles/"
|
| 176 |
+
"from-coding-novice-to-advanced-llm-developer/"
|
| 177 |
+
),
|
| 178 |
+
("$599 bundle price, was $727", "4 products"),
|
| 179 |
+
),
|
| 180 |
+
(
|
| 181 |
+
(
|
| 182 |
+
"https://towardsai.com/academy/bundles/"
|
| 183 |
+
"10-hour-crash-course-into-llm-developer-expert/"
|
| 184 |
+
),
|
| 185 |
+
("$948 bundle price, was $1,078", "4 products"),
|
| 186 |
+
),
|
| 187 |
+
(
|
| 188 |
+
"https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 189 |
+
("free preview · 7 full lessons",),
|
| 190 |
+
),
|
| 191 |
+
(
|
| 192 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 193 |
+
("free preview lessons · no card required", "instant access"),
|
| 194 |
+
),
|
| 195 |
+
],
|
| 196 |
+
)
|
| 197 |
+
def test_current_course_and_bundle_facts_are_present(
|
| 198 |
+
url: str, facts: tuple[str, ...]
|
| 199 |
+
) -> None:
|
| 200 |
+
text = _page_text(url)
|
| 201 |
+
|
| 202 |
+
for fact in facts:
|
| 203 |
+
assert fact in text
|
| 204 |
+
|
| 205 |
+
|
| 206 |
+
@pytest.mark.parametrize(
|
| 207 |
+
("query", "expected_url"),
|
| 208 |
+
[
|
| 209 |
+
(
|
| 210 |
+
"What is included in the Full Stack AI Engineering course and what does it cost?",
|
| 211 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 212 |
+
),
|
| 213 |
+
(
|
| 214 |
+
"How many lessons and what price is the Agent Engineering course?",
|
| 215 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 216 |
+
),
|
| 217 |
+
(
|
| 218 |
+
"What is the 10-Hour LLM Fundamentals course price?",
|
| 219 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 220 |
+
),
|
| 221 |
+
(
|
| 222 |
+
"Is Beginner Python for AI Engineering for non-coders?",
|
| 223 |
+
"https://towardsai.com/academy/python-for-ai-engineering/",
|
| 224 |
+
),
|
| 225 |
+
(
|
| 226 |
+
"What is Master AI for Work for professionals?",
|
| 227 |
+
"https://towardsai.com/academy/ai-for-work/",
|
| 228 |
+
),
|
| 229 |
+
(
|
| 230 |
+
"What is the Building LLMs for Production course and ebook?",
|
| 231 |
+
"https://towardsai.com/academy/building-llms-for-production/",
|
| 232 |
+
),
|
| 233 |
+
(
|
| 234 |
+
"Where are the Building LLMs book companion resources?",
|
| 235 |
+
"https://towardsai.com/academy/book/",
|
| 236 |
+
),
|
| 237 |
+
(
|
| 238 |
+
"What course is included in mentorship?",
|
| 239 |
+
"https://towardsai.com/academy/mentorship/",
|
| 240 |
+
),
|
| 241 |
+
(
|
| 242 |
+
"What does the Get It All bundle include?",
|
| 243 |
+
"https://towardsai.com/academy/bundles/get-it-all/",
|
| 244 |
+
),
|
| 245 |
+
(
|
| 246 |
+
"What is in the From Non-Coder to AI Engineer bundle?",
|
| 247 |
+
(
|
| 248 |
+
"https://towardsai.com/academy/bundles/"
|
| 249 |
+
"from-coding-novice-to-advanced-llm-developer/"
|
| 250 |
+
),
|
| 251 |
+
),
|
| 252 |
+
(
|
| 253 |
+
"What is in the From Developer to Advanced AI Engineer bundle?",
|
| 254 |
+
(
|
| 255 |
+
"https://towardsai.com/academy/bundles/"
|
| 256 |
+
"10-hour-crash-course-into-llm-developer-expert/"
|
| 257 |
+
),
|
| 258 |
+
),
|
| 259 |
+
(
|
| 260 |
+
"Where is the Full Stack free course preview?",
|
| 261 |
+
"https://towardsai.com/academy/full-stack-ai-engineering-free-preview/",
|
| 262 |
+
),
|
| 263 |
+
(
|
| 264 |
+
"How many Agent Engineering free preview lessons are there?",
|
| 265 |
+
"https://towardsai.com/academy/agent-engineering-free-preview/",
|
| 266 |
+
),
|
| 267 |
+
(
|
| 268 |
+
"Where is the free Agent Engineering webinar?",
|
| 269 |
+
"https://towardsai.com/webinars/agentengineering/",
|
| 270 |
+
),
|
| 271 |
+
(
|
| 272 |
+
"Where can I get the free agents cheatsheet?",
|
| 273 |
+
(
|
| 274 |
+
"https://academy.towardsai.net/products/"
|
| 275 |
+
"digital_downloads/agents-cheatsheet"
|
| 276 |
+
),
|
| 277 |
+
),
|
| 278 |
+
(
|
| 279 |
+
"Where can I get the Anti-Slop Framework?",
|
| 280 |
+
(
|
| 281 |
+
"https://academy.towardsai.net/products/"
|
| 282 |
+
"digital_downloads/anti-slop-framework"
|
| 283 |
+
),
|
| 284 |
+
),
|
| 285 |
+
(
|
| 286 |
+
"What enterprise AI enablement and training do you offer?",
|
| 287 |
+
"https://towardsai.com/enterpriseenablement/",
|
| 288 |
+
),
|
| 289 |
+
(
|
| 290 |
+
"Can you convert software developers into AI engineers?",
|
| 291 |
+
("https://towardsai.com/enterprise/software-developer-to-ai-engineer/"),
|
| 292 |
+
),
|
| 293 |
+
(
|
| 294 |
+
"Do you offer Claude Code and Codex training?",
|
| 295 |
+
("https://towardsai.com/enterprise/agentic-developer-conversion/"),
|
| 296 |
+
),
|
| 297 |
+
(
|
| 298 |
+
"Do you do AI deployment value creation consulting?",
|
| 299 |
+
"https://towardsai.com/valuecreation/",
|
| 300 |
+
),
|
| 301 |
+
],
|
| 302 |
+
)
|
| 303 |
+
def test_every_current_offer_routes_to_its_canonical_page(
|
| 304 |
+
query: str, expected_url: str
|
| 305 |
+
) -> None:
|
| 306 |
+
selected = retrieve(query)
|
| 307 |
+
|
| 308 |
+
assert selected, query
|
| 309 |
+
assert selected[0]["url"] == expected_url
|
| 310 |
+
|
| 311 |
+
|
| 312 |
+
def test_course_decider_starter_returns_one_canonical_hero_per_course() -> None:
|
| 313 |
+
selected = retrieve("I want help deciding which course to take.")
|
| 314 |
+
|
| 315 |
+
assert len(selected) == 6
|
| 316 |
+
assert len({page["url"] for page in selected}) == 6
|
| 317 |
+
assert all(page["chunk_index"] == 0 for page in selected)
|
| 318 |
+
assert {page["path"] for page in selected} == {
|
| 319 |
+
"/academy/full-stack-ai-engineering",
|
| 320 |
+
"/academy/agent-engineering",
|
| 321 |
+
"/academy/llm-primer",
|
| 322 |
+
"/academy/python-for-ai-engineering",
|
| 323 |
+
"/academy/ai-for-work",
|
| 324 |
+
"/academy/building-llms-for-production",
|
| 325 |
+
}
|
| 326 |
+
assert len(llm.evidence_chunks(selected)) == len(selected)
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
def test_company_starters_return_only_canonical_b2b_evidence() -> None:
|
| 330 |
+
integration = retrieve("I want help to integrate AI into my company")
|
| 331 |
+
training = retrieve("I want a training inside my company")
|
| 332 |
+
|
| 333 |
+
assert integration[0]["path"] == "/valuecreation"
|
| 334 |
+
assert integration[1]["path"] == "/enterpriseenablement"
|
| 335 |
+
assert all(page["kind"] == "b2b" for page in integration)
|
| 336 |
+
assert training[0]["path"] == "/enterpriseenablement"
|
| 337 |
+
assert all(page["kind"] == "b2b" for page in training)
|
| 338 |
+
|
| 339 |
+
|
| 340 |
+
def test_free_resource_starter_does_not_retrieve_paid_offer_evidence() -> None:
|
| 341 |
+
selected = retrieve(
|
| 342 |
+
"I'm looking for more free resources to learn before committing to buying "
|
| 343 |
+
"a course"
|
| 344 |
+
)
|
| 345 |
+
|
| 346 |
+
assert selected
|
| 347 |
+
assert selected[0]["kind"] == "free_resource"
|
| 348 |
+
assert all(
|
| 349 |
+
page["kind"] in {"free_resource", "digital_download"}
|
| 350 |
+
or page["path"] == "/academy/book"
|
| 351 |
+
for page in selected
|
| 352 |
+
)
|
| 353 |
+
assert len(llm.evidence_chunks(selected)) == len(selected)
|
| 354 |
+
|
| 355 |
+
|
| 356 |
+
def test_mentor_starter_returns_only_mentorship_evidence() -> None:
|
| 357 |
+
selected = retrieve("I want to find mentors")
|
| 358 |
+
|
| 359 |
+
assert selected
|
| 360 |
+
assert all(page["path"] == "/academy/mentorship" for page in selected)
|
| 361 |
+
|
| 362 |
+
|
| 363 |
+
def test_canonical_offer_pages_suppress_lower_authority_mirrors() -> None:
|
| 364 |
+
com_offer_ids = {
|
| 365 |
+
page["offer_id"]
|
| 366 |
+
for page in _catalog("towardsai_com_pages.json")["pages"]
|
| 367 |
+
if page.get("offer_id")
|
| 368 |
+
}
|
| 369 |
+
|
| 370 |
+
for page in pages():
|
| 371 |
+
if page.get("offer_id") in com_offer_ids:
|
| 372 |
+
assert page["authority"] == "official_site"
|
| 373 |
+
|
| 374 |
+
|
| 375 |
+
def test_mentorship_incident_is_answered_only_from_exact_current_evidence() -> None:
|
| 376 |
+
selected = retrieve(
|
| 377 |
+
"The chatbot mentioned access to 2 courses of our choice as part of "
|
| 378 |
+
"mentorship. Is that true?",
|
| 379 |
+
current_url="https://towardsai.com/academy/mentorship/",
|
| 380 |
+
)
|
| 381 |
+
|
| 382 |
+
assert selected[0]["url"] == "https://towardsai.com/academy/mentorship/"
|
| 383 |
+
assert "Included from day one" in selected[0]["text"]
|
| 384 |
+
quote = (
|
| 385 |
+
"10-Hour LLM Fundamentals video course Five in-depth 2-hour video "
|
| 386 |
+
"sessions, from a basic prompt to a full production rollout $199 "
|
| 387 |
+
"Included from day one"
|
| 388 |
+
)
|
| 389 |
+
claim = f"{quote}."
|
| 390 |
+
raw = json.dumps(
|
| 391 |
+
{
|
| 392 |
+
"status": "answered",
|
| 393 |
+
"claims": [
|
| 394 |
+
{
|
| 395 |
+
"text": claim,
|
| 396 |
+
"chunk_id": selected[0]["chunk_id"],
|
| 397 |
+
"quote": quote,
|
| 398 |
+
}
|
| 399 |
+
],
|
| 400 |
+
}
|
| 401 |
+
)
|
| 402 |
+
|
| 403 |
+
result = llm.validate_grounded_result(raw, selected)
|
| 404 |
+
|
| 405 |
+
assert result.is_answered
|
| 406 |
+
assert result.answer == claim
|
| 407 |
+
assert "two courses" not in result.answer.lower()
|
| 408 |
+
|
| 409 |
+
|
| 410 |
+
def test_no_active_chunk_contains_the_retired_two_course_offer() -> None:
|
| 411 |
+
retired_phrases = (
|
| 412 |
+
"two courses included",
|
| 413 |
+
"pick one from these options",
|
| 414 |
+
"ai coding: a fixed course",
|
| 415 |
+
)
|
| 416 |
+
active_text = "\n".join(
|
| 417 |
+
chunk["text"] for page in pages() for chunk in page["chunks"]
|
| 418 |
+
).lower()
|
| 419 |
+
|
| 420 |
+
assert not any(phrase in active_text for phrase in retired_phrases)
|
| 421 |
+
assert not any(
|
| 422 |
+
page["url"] == "https://towardsai.com/academy/membership/" for page in pages()
|
| 423 |
+
)
|
| 424 |
+
|
| 425 |
+
|
| 426 |
+
def test_stale_academy_summaries_are_not_retrieval_evidence() -> None:
|
| 427 |
+
blocked_urls = {
|
| 428 |
+
"https://towardsai.com/",
|
| 429 |
+
"https://towardsai.com/academy/",
|
| 430 |
+
"https://towardsai.com/academy/bundles/",
|
| 431 |
+
"https://academy.towardsai.net/collections",
|
| 432 |
+
"https://academy.towardsai.net/collections/products",
|
| 433 |
+
"https://academy.towardsai.net/collections/developers",
|
| 434 |
+
"https://academy.towardsai.net/collections/professionals",
|
| 435 |
+
"https://academy.towardsai.net/pages/free-resources",
|
| 436 |
+
}
|
| 437 |
+
active_pages = pages()
|
| 438 |
+
active_urls = {page["url"] for page in active_pages}
|
| 439 |
+
active_text = "\n".join(page["text"] for page in active_pages).lower()
|
| 440 |
+
|
| 441 |
+
assert blocked_urls.isdisjoint(active_urls)
|
| 442 |
+
assert "124 lessons" not in active_text
|
| 443 |
+
assert "free resources from 8-hour llm primer" not in active_text
|
| 444 |
+
assert "465-page reference" not in active_text
|
| 445 |
+
|
| 446 |
+
|
| 447 |
+
def test_navigation_and_announcement_boilerplate_is_not_evidence() -> None:
|
| 448 |
+
active_text = "\n".join(page["text"] for page in pages()).lower()
|
| 449 |
+
|
| 450 |
+
assert "new: towards ai mentorship" not in active_text
|
| 451 |
+
assert "skip to main content" not in active_text
|
| 452 |
+
assert "toggle menu" not in active_text
|
| 453 |
+
|
| 454 |
+
|
| 455 |
+
@pytest.mark.parametrize(
|
| 456 |
+
("query", "expected_url", "retired_fact"),
|
| 457 |
+
[
|
| 458 |
+
(
|
| 459 |
+
"How many lessons are in Full Stack AI Engineering?",
|
| 460 |
+
"https://towardsai.com/academy/full-stack-ai-engineering/",
|
| 461 |
+
"124 lessons",
|
| 462 |
+
),
|
| 463 |
+
(
|
| 464 |
+
"How many lessons are in Agent Engineering?",
|
| 465 |
+
"https://towardsai.com/academy/agent-engineering/",
|
| 466 |
+
"50 lessons",
|
| 467 |
+
),
|
| 468 |
+
(
|
| 469 |
+
"Do you have an 8-hour LLM Primer?",
|
| 470 |
+
"https://towardsai.com/academy/llm-primer/",
|
| 471 |
+
"8-hour llm primer",
|
| 472 |
+
),
|
| 473 |
+
],
|
| 474 |
+
)
|
| 475 |
+
def test_named_offer_queries_use_current_canonical_evidence(
|
| 476 |
+
query: str, expected_url: str, retired_fact: str
|
| 477 |
+
) -> None:
|
| 478 |
+
selected = retrieve(query)
|
| 479 |
+
|
| 480 |
+
assert selected
|
| 481 |
+
assert selected[0]["url"] == expected_url
|
| 482 |
+
assert retired_fact not in "\n".join(page["text"] for page in selected).lower()
|
| 483 |
+
|
| 484 |
+
|
| 485 |
+
def test_book_page_count_is_an_answerable_atomic_span() -> None:
|
| 486 |
+
selected = retrieve("How many pages is Building LLMs for Production?")
|
| 487 |
+
|
| 488 |
+
assert selected
|
| 489 |
+
assert selected[0]["url"] == (
|
| 490 |
+
"https://towardsai.com/academy/building-llms-for-production/"
|
| 491 |
+
)
|
| 492 |
+
assert any("470-page" in span["text"] for span in selected[0]["evidence_spans"])
|
| 493 |
+
|
| 494 |
+
|
| 495 |
+
@pytest.mark.parametrize(
|
| 496 |
+
"query",
|
| 497 |
+
[
|
| 498 |
+
"Does Agent Engineering free preview include lifetime access and require no card?",
|
| 499 |
+
"Does the Full Stack free preview give me a certificate?",
|
| 500 |
+
"Does Building LLMs for Production have a refund guarantee?",
|
| 501 |
+
"How long is Agent Engineering?",
|
| 502 |
+
"How many hours does Python for AI Engineering take?",
|
| 503 |
+
"How many lessons are in LLM Fundamentals?",
|
| 504 |
+
"Does mentorship include a certificate?",
|
| 505 |
+
"Does the Get It All bundle get 50% off?",
|
| 506 |
+
],
|
| 507 |
+
)
|
| 508 |
+
def test_unpublished_high_risk_offer_facts_have_no_citable_evidence(
|
| 509 |
+
query: str,
|
| 510 |
+
) -> None:
|
| 511 |
+
assert retrieve(query) == []
|
| 512 |
+
|
| 513 |
+
|
| 514 |
+
def test_named_llm_fundamentals_feature_query_cannot_retrieve_another_offer() -> None:
|
| 515 |
+
selected = retrieve(
|
| 516 |
+
"Does LLM Fundamentals include community support and an AI tutor?"
|
| 517 |
+
)
|
| 518 |
+
|
| 519 |
+
assert selected
|
| 520 |
+
assert {chunk["offer_id"] for chunk in selected} == {"llm-primer"}
|
| 521 |
+
assert any(
|
| 522 |
+
"community" in span["text"].casefold()
|
| 523 |
+
and "ai tutor" in span["text"].casefold()
|
| 524 |
+
for chunk in selected
|
| 525 |
+
for span in chunk["evidence_spans"]
|
| 526 |
+
)
|
| 527 |
+
|
| 528 |
+
|
| 529 |
+
def test_get_it_all_exact_price_abstains_when_only_checkout_policy_is_published() -> None:
|
| 530 |
+
assert retrieve("What is the Get It All bundle price?") == []
|
| 531 |
+
|
| 532 |
+
location = retrieve("Where is the Get It All bundle price shown at checkout?")
|
| 533 |
+
spans = [
|
| 534 |
+
span["text"]
|
| 535 |
+
for chunk in location
|
| 536 |
+
for span in chunk.get("evidence_spans", [])
|
| 537 |
+
]
|
| 538 |
+
assert location
|
| 539 |
+
assert {chunk["offer_id"] for chunk in location} == {"get-it-all"}
|
| 540 |
+
assert spans
|
| 541 |
+
assert all("Bundle price shown at checkout" in span for span in spans)
|
| 542 |
+
assert all("$1,625" not in span and "$948" not in span for span in spans)
|
| 543 |
+
|
| 544 |
+
|
| 545 |
+
def test_full_stack_preview_count_uses_only_explicit_preview_count_evidence() -> None:
|
| 546 |
+
selected = retrieve("How many Full Stack free preview lessons are there?")
|
| 547 |
+
spans = [
|
| 548 |
+
span["text"]
|
| 549 |
+
for chunk in selected
|
| 550 |
+
for span in chunk.get("evidence_spans", [])
|
| 551 |
+
]
|
| 552 |
+
|
| 553 |
+
assert selected
|
| 554 |
+
assert {chunk["offer_id"] for chunk in selected} <= {
|
| 555 |
+
"full-stack-ai-engineering-free-preview",
|
| 556 |
+
"full-stack-ai-engineering",
|
| 557 |
+
}
|
| 558 |
+
assert spans
|
| 559 |
+
assert all("6" in span or "six" in span.casefold() for span in spans)
|
| 560 |
+
assert all("92 lessons" not in span for span in spans)
|
| 561 |
+
|
| 562 |
+
|
| 563 |
+
def test_preview_count_exception_never_exposes_paid_access_evidence() -> None:
|
| 564 |
+
selected = retrieve("How many Agent Engineering preview lessons are there?")
|
| 565 |
+
evidence = [
|
| 566 |
+
span["text"].casefold()
|
| 567 |
+
for chunk in selected
|
| 568 |
+
for span in chunk["evidence_spans"]
|
| 569 |
+
]
|
| 570 |
+
|
| 571 |
+
assert selected
|
| 572 |
+
assert any("7" in span and "lesson" in span for span in evidence)
|
| 573 |
+
assert all("lifetime access" not in span for span in evidence)
|
| 574 |
+
|
| 575 |
+
|
| 576 |
+
@pytest.mark.parametrize(
|
| 577 |
+
"query",
|
| 578 |
+
[
|
| 579 |
+
(
|
| 580 |
+
"How many Agent Engineering preview lessons are there, and what does "
|
| 581 |
+
"the full course cost?"
|
| 582 |
+
),
|
| 583 |
+
(
|
| 584 |
+
"How many Full Stack preview lessons are there, and does the full "
|
| 585 |
+
"course include lifetime access?"
|
| 586 |
+
),
|
| 587 |
+
],
|
| 588 |
+
)
|
| 589 |
+
def test_mixed_preview_and_paid_fact_questions_fail_closed(query: str) -> None:
|
| 590 |
+
assert retrieve(query) == []
|
| 591 |
+
|
| 592 |
+
|
| 593 |
+
def test_lifetime_access_requires_an_explicit_permanent_access_qualifier() -> None:
|
| 594 |
+
selected = retrieve("Does the Full Stack free preview include lifetime access?")
|
| 595 |
+
evidence = [
|
| 596 |
+
span["text"].casefold()
|
| 597 |
+
for chunk in selected
|
| 598 |
+
for span in chunk["evidence_spans"]
|
| 599 |
+
]
|
| 600 |
+
|
| 601 |
+
assert selected
|
| 602 |
+
assert evidence
|
| 603 |
+
assert all(
|
| 604 |
+
any(term in span for term in ("lifetime", "forever", "keep", "retain"))
|
| 605 |
+
for span in evidence
|
| 606 |
+
)
|
| 607 |
+
|
| 608 |
+
|
| 609 |
+
def test_mentorship_cancellation_retrieves_only_the_exact_access_policy() -> None:
|
| 610 |
+
selected = retrieve(
|
| 611 |
+
"Do I keep LLM Fundamentals lifetime access after cancelling mentorship?"
|
| 612 |
+
)
|
| 613 |
+
spans = {
|
| 614 |
+
span["text"]
|
| 615 |
+
for chunk in selected
|
| 616 |
+
for span in chunk.get("evidence_spans", [])
|
| 617 |
+
}
|
| 618 |
+
|
| 619 |
+
assert selected
|
| 620 |
+
assert {chunk["offer_id"] for chunk in selected} == {"mentorship"}
|
| 621 |
+
assert any("Course access is active while" in span for span in spans)
|
| 622 |
+
assert any("If you cancel" in span and "lifetime access" in span for span in spans)
|
| 623 |
+
assert all("Included from day one" not in span for span in spans)
|
| 624 |
+
|
| 625 |
+
|
| 626 |
+
@pytest.mark.parametrize(
|
| 627 |
+
"query",
|
| 628 |
+
[
|
| 629 |
+
"Does the monthly mentorship plan have a 30-day money-back guarantee?",
|
| 630 |
+
"Does the month-to-month mentorship have a 30-day money-back guarantee?",
|
| 631 |
+
],
|
| 632 |
+
)
|
| 633 |
+
def test_monthly_mentorship_guarantee_retrieves_both_plan_qualifiers(
|
| 634 |
+
query: str,
|
| 635 |
+
) -> None:
|
| 636 |
+
selected = retrieve(query)
|
| 637 |
+
spans = {
|
| 638 |
+
span["text"]
|
| 639 |
+
for chunk in selected
|
| 640 |
+
for span in chunk.get("evidence_spans", [])
|
| 641 |
+
}
|
| 642 |
+
|
| 643 |
+
assert selected
|
| 644 |
+
assert any("monthly mentorship" in span.casefold() for span in spans)
|
| 645 |
+
assert any(
|
| 646 |
+
"yearly" in span.casefold() and "money-back" in span.casefold()
|
| 647 |
+
for span in spans
|
| 648 |
+
)
|
| 649 |
+
assert all("Join the Mentorship 30-day" not in span for span in spans)
|
| 650 |
+
|
| 651 |
+
|
| 652 |
+
def test_mentorship_plan_prices_require_exact_plan_denominations() -> None:
|
| 653 |
+
monthly_queries = (
|
| 654 |
+
"What is the monthly mentorship price?",
|
| 655 |
+
"Mentorship month price?",
|
| 656 |
+
"What is the month-to-month mentorship price?",
|
| 657 |
+
"What is the mentorship price per month?",
|
| 658 |
+
"What does mentorship cost each month?",
|
| 659 |
+
)
|
| 660 |
+
monthly_results = [
|
| 661 |
+
[
|
| 662 |
+
span["text"]
|
| 663 |
+
for chunk in retrieve(query)
|
| 664 |
+
for span in chunk["evidence_spans"]
|
| 665 |
+
]
|
| 666 |
+
for query in monthly_queries
|
| 667 |
+
]
|
| 668 |
+
yearly_queries = (
|
| 669 |
+
"What is the yearly mentorship price?",
|
| 670 |
+
"Mentorship year price?",
|
| 671 |
+
"What is the annual mentorship price?",
|
| 672 |
+
"What is the mentorship price per year?",
|
| 673 |
+
)
|
| 674 |
+
yearly_results = [
|
| 675 |
+
[
|
| 676 |
+
span["text"]
|
| 677 |
+
for chunk in retrieve(query)
|
| 678 |
+
for span in chunk["evidence_spans"]
|
| 679 |
+
]
|
| 680 |
+
for query in yearly_queries
|
| 681 |
+
]
|
| 682 |
+
|
| 683 |
+
assert all(monthly_results)
|
| 684 |
+
for monthly in monthly_results:
|
| 685 |
+
assert all("Guest workshop monthly" not in span for span in monthly)
|
| 686 |
+
assert all("Production blueprint monthly" not in span for span in monthly)
|
| 687 |
+
assert all("From $75/month" not in span for span in monthly)
|
| 688 |
+
assert any("$99" in span for span in monthly)
|
| 689 |
+
assert all(yearly_results)
|
| 690 |
+
for yearly in yearly_results:
|
| 691 |
+
assert all("$289 off" not in span for span in yearly)
|
| 692 |
+
assert all("$75 a month billed yearly" not in span for span in yearly)
|
| 693 |
+
assert any("$899" in span and "/year" in span for span in yearly)
|
| 694 |
+
|
| 695 |
+
|
| 696 |
+
def test_mentorship_discounts_are_bound_to_the_requested_plan_or_courses() -> None:
|
| 697 |
+
yearly = [
|
| 698 |
+
span["text"]
|
| 699 |
+
for chunk in retrieve("Is the yearly mentorship plan 24% off?")
|
| 700 |
+
for span in chunk["evidence_spans"]
|
| 701 |
+
]
|
| 702 |
+
saved = [
|
| 703 |
+
span["text"]
|
| 704 |
+
for chunk in retrieve("Does mentorship save 24%?")
|
| 705 |
+
for span in chunk["evidence_spans"]
|
| 706 |
+
]
|
| 707 |
+
courses = [
|
| 708 |
+
span["text"]
|
| 709 |
+
for chunk in retrieve(
|
| 710 |
+
"What discount does mentorship give on Full Stack AI Engineering?"
|
| 711 |
+
)
|
| 712 |
+
for span in chunk["evidence_spans"]
|
| 713 |
+
]
|
| 714 |
+
|
| 715 |
+
assert yearly
|
| 716 |
+
assert all(
|
| 717 |
+
"save 24%" in span.casefold() or "24% less" in span.casefold()
|
| 718 |
+
for span in yearly
|
| 719 |
+
)
|
| 720 |
+
assert all("25% off" not in span for span in yearly)
|
| 721 |
+
assert saved
|
| 722 |
+
assert all(
|
| 723 |
+
"save 24%" in span.casefold() or "24% less" in span.casefold()
|
| 724 |
+
for span in saved
|
| 725 |
+
)
|
| 726 |
+
assert courses
|
| 727 |
+
assert all("25% off" in span for span in courses)
|
| 728 |
+
assert retrieve("Does monthly mentorship have a discount?") == []
|
| 729 |
+
|
| 730 |
+
|
| 731 |
+
def test_mentorship_access_and_features_exclude_comparison_entities() -> None:
|
| 732 |
+
access = [
|
| 733 |
+
span["text"]
|
| 734 |
+
for chunk in retrieve("Does mentorship include lifetime course access?")
|
| 735 |
+
for span in chunk["evidence_spans"]
|
| 736 |
+
]
|
| 737 |
+
reviews = [
|
| 738 |
+
span["text"]
|
| 739 |
+
for chunk in retrieve("Does mentorship include resume and project reviews?")
|
| 740 |
+
for span in chunk["evidence_spans"]
|
| 741 |
+
]
|
| 742 |
+
|
| 743 |
+
assert access
|
| 744 |
+
assert all("retainers" not in span.casefold() for span in access)
|
| 745 |
+
assert any("lifetime access" in span.casefold() for span in access)
|
| 746 |
+
assert reviews
|
| 747 |
+
assert any(
|
| 748 |
+
"resume" in span.casefold()
|
| 749 |
+
and "project" in span.casefold()
|
| 750 |
+
and "reviews" in span.casefold()
|
| 751 |
+
for span in reviews
|
| 752 |
+
)
|
| 753 |
+
assert all("single mentor" not in span.casefold() for span in reviews)
|
tests/test_retrieval_safety.py
ADDED
|
@@ -0,0 +1,414 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import hashlib
|
| 4 |
+
import json
|
| 5 |
+
from datetime import UTC, datetime, timedelta
|
| 6 |
+
from typing import Any
|
| 7 |
+
|
| 8 |
+
import pytest
|
| 9 |
+
from fastapi.testclient import TestClient
|
| 10 |
+
|
| 11 |
+
from tai_helper import api, catalog, llm
|
| 12 |
+
from tai_helper.schemas import HelperChatRequest
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def _evidence_hash(page: dict[str, Any]) -> str:
|
| 16 |
+
payload = json.dumps(
|
| 17 |
+
{key: value for key, value in page.items() if key != "evidence_hash"},
|
| 18 |
+
ensure_ascii=False,
|
| 19 |
+
sort_keys=True,
|
| 20 |
+
separators=(",", ":"),
|
| 21 |
+
)
|
| 22 |
+
return hashlib.sha256(payload.encode()).hexdigest()
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def _page(
|
| 26 |
+
*,
|
| 27 |
+
slug: str = "agent-engineering",
|
| 28 |
+
text: str = "Build production-ready AI agents with practical engineering lessons.",
|
| 29 |
+
fetched_at: str | None = None,
|
| 30 |
+
**overrides: Any,
|
| 31 |
+
) -> dict[str, Any]:
|
| 32 |
+
fetched_at = fetched_at or datetime.now(UTC).isoformat()
|
| 33 |
+
page: dict[str, Any] = {
|
| 34 |
+
"url": f"https://towardsai.com/academy/{slug}/",
|
| 35 |
+
"canonical_url": f"https://towardsai.com/academy/{slug}/",
|
| 36 |
+
"host": "towardsai.com",
|
| 37 |
+
"path": f"/academy/{slug}",
|
| 38 |
+
"title": slug.replace("-", " ").title(),
|
| 39 |
+
"kind": "course",
|
| 40 |
+
"offer_id": slug,
|
| 41 |
+
"entity_id": f"offer:{slug}",
|
| 42 |
+
"status": "included",
|
| 43 |
+
"retrieval_eligible": True,
|
| 44 |
+
"authority": "canonical_offer",
|
| 45 |
+
"fetched_at": fetched_at,
|
| 46 |
+
"content_hash": hashlib.sha256(text.encode()).hexdigest(),
|
| 47 |
+
"http_status": 200,
|
| 48 |
+
"text": text,
|
| 49 |
+
"chunks": [
|
| 50 |
+
{
|
| 51 |
+
"chunk_id": f"{slug}:overview",
|
| 52 |
+
"heading": "Overview",
|
| 53 |
+
"text": text,
|
| 54 |
+
"evidence_spans": [
|
| 55 |
+
{"span_id": f"{slug}:overview:span-0", "text": text}
|
| 56 |
+
],
|
| 57 |
+
}
|
| 58 |
+
],
|
| 59 |
+
}
|
| 60 |
+
page["evidence_hash"] = _evidence_hash(page)
|
| 61 |
+
page.update(overrides)
|
| 62 |
+
return page
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
@pytest.fixture(autouse=True)
|
| 66 |
+
def clear_caches_between_tests() -> None:
|
| 67 |
+
catalog.clear_catalog_caches()
|
| 68 |
+
yield
|
| 69 |
+
catalog.clear_catalog_caches()
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
@pytest.fixture
|
| 73 |
+
def install_catalog(monkeypatch: pytest.MonkeyPatch):
|
| 74 |
+
def install(*pages: dict[str, Any]) -> None:
|
| 75 |
+
monkeypatch.setattr(
|
| 76 |
+
catalog,
|
| 77 |
+
"pages_payload",
|
| 78 |
+
lambda: {"pages": list(pages), "generated_at": {}},
|
| 79 |
+
)
|
| 80 |
+
catalog.clear_catalog_caches()
|
| 81 |
+
|
| 82 |
+
return install
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
@pytest.mark.parametrize(
|
| 86 |
+
("mutation", "value"),
|
| 87 |
+
[
|
| 88 |
+
("remove", "status"),
|
| 89 |
+
("set_status", "review_pending"),
|
| 90 |
+
],
|
| 91 |
+
ids=["missing-status", "unknown-status"],
|
| 92 |
+
)
|
| 93 |
+
def test_only_explicit_active_status_can_supply_evidence(
|
| 94 |
+
install_catalog, mutation: str, value: str
|
| 95 |
+
) -> None:
|
| 96 |
+
valid = _page(slug="valid-course", text="Valid course teaches glacier models.")
|
| 97 |
+
invalid = _page(slug="invalid-course", text="Invalid course teaches zeppelins.")
|
| 98 |
+
if mutation == "remove":
|
| 99 |
+
invalid.pop(value)
|
| 100 |
+
else:
|
| 101 |
+
invalid["status"] = value
|
| 102 |
+
install_catalog(valid, invalid)
|
| 103 |
+
|
| 104 |
+
assert [page["url"] for page in catalog.pages()] == [valid["url"]]
|
| 105 |
+
assert catalog.retrieve("zeppelins") == []
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
@pytest.mark.parametrize("eligible", [None, False], ids=["missing", "false"])
|
| 109 |
+
def test_retrieval_eligibility_must_be_explicitly_true(
|
| 110 |
+
install_catalog, eligible: bool | None
|
| 111 |
+
) -> None:
|
| 112 |
+
page = _page(text="A course about orbital pottery.")
|
| 113 |
+
if eligible is None:
|
| 114 |
+
page.pop("retrieval_eligible")
|
| 115 |
+
else:
|
| 116 |
+
page["retrieval_eligible"] = eligible
|
| 117 |
+
install_catalog(page)
|
| 118 |
+
|
| 119 |
+
assert catalog.pages() == []
|
| 120 |
+
assert catalog.retrieve("orbital pottery course") == []
|
| 121 |
+
|
| 122 |
+
|
| 123 |
+
def test_content_hash_is_recomputed_before_page_can_supply_evidence(
|
| 124 |
+
install_catalog,
|
| 125 |
+
) -> None:
|
| 126 |
+
page = _page(text="Original verified course content.")
|
| 127 |
+
page["text"] = "Tampered course content."
|
| 128 |
+
install_catalog(page)
|
| 129 |
+
|
| 130 |
+
assert catalog.pages() == []
|
| 131 |
+
assert catalog.retrieve("tampered course content") == []
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
@pytest.mark.parametrize(
|
| 135 |
+
("field", "value"),
|
| 136 |
+
[
|
| 137 |
+
("authority", "primary"),
|
| 138 |
+
("offer_id", ""),
|
| 139 |
+
("kind", "bundle"),
|
| 140 |
+
("title", "Altered trusted title"),
|
| 141 |
+
],
|
| 142 |
+
)
|
| 143 |
+
def test_evidence_hash_binds_retrieval_provenance_and_metadata(
|
| 144 |
+
install_catalog, field: str, value: str
|
| 145 |
+
) -> None:
|
| 146 |
+
page = _page(text="Hash-bound canonical course evidence.")
|
| 147 |
+
page[field] = value
|
| 148 |
+
install_catalog(page)
|
| 149 |
+
|
| 150 |
+
assert catalog.pages() == []
|
| 151 |
+
assert catalog.retrieve("canonical course evidence") == []
|
| 152 |
+
|
| 153 |
+
|
| 154 |
+
def test_chunk_text_must_belong_to_the_hashed_page_text(install_catalog) -> None:
|
| 155 |
+
page = _page(text="The verified page includes one fundamentals course.")
|
| 156 |
+
forged = "The mentorship includes two courses of your choice."
|
| 157 |
+
page["chunks"][0]["text"] = forged
|
| 158 |
+
page["chunks"][0]["evidence_spans"] = [{"span_id": "forged:span-0", "text": forged}]
|
| 159 |
+
install_catalog(page)
|
| 160 |
+
|
| 161 |
+
assert catalog.pages() == []
|
| 162 |
+
assert catalog.retrieve("two courses of your choice") == []
|
| 163 |
+
|
| 164 |
+
|
| 165 |
+
@pytest.mark.parametrize(
|
| 166 |
+
("field", "value"),
|
| 167 |
+
[
|
| 168 |
+
("url", "https://evil.example/phish"),
|
| 169 |
+
("host", "evil.example"),
|
| 170 |
+
("path", "/academy/different-course"),
|
| 171 |
+
],
|
| 172 |
+
)
|
| 173 |
+
def test_citation_url_and_metadata_must_match_the_canonical_page(
|
| 174 |
+
install_catalog, field: str, value: str
|
| 175 |
+
) -> None:
|
| 176 |
+
page = _page(text="Verified canonical course evidence.")
|
| 177 |
+
page[field] = value
|
| 178 |
+
install_catalog(page)
|
| 179 |
+
|
| 180 |
+
assert catalog.pages() == []
|
| 181 |
+
assert catalog.retrieve("canonical course evidence") == []
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
def test_duplicate_chunk_ids_across_pages_fail_closed(install_catalog) -> None:
|
| 185 |
+
first = _page(slug="first", text="First unique course evidence.")
|
| 186 |
+
second = _page(slug="second", text="Second unique course evidence.")
|
| 187 |
+
second["chunks"][0]["chunk_id"] = first["chunks"][0]["chunk_id"]
|
| 188 |
+
second["evidence_hash"] = _evidence_hash(second)
|
| 189 |
+
install_catalog(first, second)
|
| 190 |
+
|
| 191 |
+
assert catalog.pages() == []
|
| 192 |
+
assert catalog.retrieve("unique course evidence") == []
|
| 193 |
+
|
| 194 |
+
|
| 195 |
+
@pytest.mark.parametrize("mutation", ["missing", "not_in_chunk"])
|
| 196 |
+
def test_atomic_evidence_spans_are_required_and_verified(
|
| 197 |
+
install_catalog, mutation: str
|
| 198 |
+
) -> None:
|
| 199 |
+
page = _page(text="No code required for this course.")
|
| 200 |
+
if mutation == "missing":
|
| 201 |
+
page["chunks"][0].pop("evidence_spans")
|
| 202 |
+
else:
|
| 203 |
+
page["chunks"][0]["evidence_spans"][0]["text"] = "Invented span."
|
| 204 |
+
install_catalog(page)
|
| 205 |
+
|
| 206 |
+
assert catalog.pages() == []
|
| 207 |
+
assert catalog.retrieve("code required") == []
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
@pytest.mark.parametrize(
|
| 211 |
+
"fetched_at",
|
| 212 |
+
[
|
| 213 |
+
None,
|
| 214 |
+
(
|
| 215 |
+
datetime.now(UTC)
|
| 216 |
+
- timedelta(days=max(catalog.settings.catalog_max_age_days, 0) + 1)
|
| 217 |
+
).isoformat(),
|
| 218 |
+
(datetime.now(UTC) + timedelta(minutes=10)).isoformat(),
|
| 219 |
+
],
|
| 220 |
+
ids=["missing", "stale", "future"],
|
| 221 |
+
)
|
| 222 |
+
def test_missing_stale_or_future_fetch_time_cannot_supply_evidence(
|
| 223 |
+
install_catalog, fetched_at: str | None
|
| 224 |
+
) -> None:
|
| 225 |
+
page = _page(text="A course about lunar basket weaving.")
|
| 226 |
+
if fetched_at is None:
|
| 227 |
+
page.pop("fetched_at")
|
| 228 |
+
else:
|
| 229 |
+
page["fetched_at"] = fetched_at
|
| 230 |
+
install_catalog(page)
|
| 231 |
+
|
| 232 |
+
assert catalog.pages() == []
|
| 233 |
+
assert catalog.retrieve("lunar basket weaving course") == []
|
| 234 |
+
|
| 235 |
+
|
| 236 |
+
def test_corrupt_catalog_reload_discards_previously_cached_evidence(
|
| 237 |
+
monkeypatch: pytest.MonkeyPatch, tmp_path
|
| 238 |
+
) -> None:
|
| 239 |
+
data_dir = tmp_path / "data"
|
| 240 |
+
data_dir.mkdir()
|
| 241 |
+
academy_path = data_dir / "pages.json"
|
| 242 |
+
website_path = data_dir / "towardsai_com_pages.json"
|
| 243 |
+
academy_path.write_text(json.dumps({"pages": []}))
|
| 244 |
+
website_path.write_text(
|
| 245 |
+
json.dumps({"pages": [_page(text="Quantum origami course.")]})
|
| 246 |
+
)
|
| 247 |
+
monkeypatch.setattr(catalog, "repo_root", lambda: tmp_path)
|
| 248 |
+
catalog.clear_catalog_caches()
|
| 249 |
+
|
| 250 |
+
assert [page["url"] for page in catalog.pages()] == [
|
| 251 |
+
"https://towardsai.com/academy/agent-engineering/"
|
| 252 |
+
]
|
| 253 |
+
|
| 254 |
+
website_path.write_text("{ definitely not valid JSON")
|
| 255 |
+
|
| 256 |
+
assert catalog.pages() == []
|
| 257 |
+
assert catalog.retrieve("quantum origami course") == []
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
@pytest.mark.parametrize(
|
| 261 |
+
"query",
|
| 262 |
+
[
|
| 263 |
+
"Does your course include scuba-diving lessons?",
|
| 264 |
+
"Can company training certify helicopter pilots?",
|
| 265 |
+
"Does the mentorship provide veterinary surgery?",
|
| 266 |
+
],
|
| 267 |
+
)
|
| 268 |
+
def test_unsupported_but_in_scope_subjects_return_no_chunks(
|
| 269 |
+
install_catalog, query: str
|
| 270 |
+
) -> None:
|
| 271 |
+
install_catalog(
|
| 272 |
+
_page(text="This AI engineering course covers agents and deployment."),
|
| 273 |
+
)
|
| 274 |
+
|
| 275 |
+
assert catalog.in_scope(query)
|
| 276 |
+
assert catalog.retrieve(query) == []
|
| 277 |
+
|
| 278 |
+
|
| 279 |
+
def test_punctuation_and_hyphens_do_not_break_relevant_retrieval(
|
| 280 |
+
install_catalog,
|
| 281 |
+
) -> None:
|
| 282 |
+
page = _page(
|
| 283 |
+
text="The course teaches production-ready agent-engineering workflows in C++ and C#.",
|
| 284 |
+
)
|
| 285 |
+
install_catalog(page)
|
| 286 |
+
|
| 287 |
+
assert {
|
| 288 |
+
"production",
|
| 289 |
+
"ready",
|
| 290 |
+
"agent",
|
| 291 |
+
"engineering",
|
| 292 |
+
"c++",
|
| 293 |
+
"c#",
|
| 294 |
+
} <= catalog.tokenize("production-ready agent-engineering C++/C#")
|
| 295 |
+
selected = catalog.retrieve("Is it production-ready for agent engineering?")
|
| 296 |
+
|
| 297 |
+
assert selected
|
| 298 |
+
assert selected[0]["url"] == page["url"]
|
| 299 |
+
|
| 300 |
+
|
| 301 |
+
def test_scope_uses_history_only_for_referential_followups() -> None:
|
| 302 |
+
course_history = ["Tell me about the Towards AI mentorship and its courses."]
|
| 303 |
+
|
| 304 |
+
assert catalog.in_scope("Is it included?", course_history)
|
| 305 |
+
assert not catalog.in_scope("Who won the World Cup?", course_history)
|
| 306 |
+
assert not catalog.in_scope("Is it included?", ["Who won the World Cup?"])
|
| 307 |
+
|
| 308 |
+
|
| 309 |
+
def test_retrieval_query_never_uses_prior_assistant_claims_as_evidence() -> None:
|
| 310 |
+
payload = HelperChatRequest.model_validate(
|
| 311 |
+
{
|
| 312 |
+
"query": "Is that accurate?",
|
| 313 |
+
"selectedPrompt": "I want to find mentors",
|
| 314 |
+
"history": [
|
| 315 |
+
{
|
| 316 |
+
"role": "assistant",
|
| 317 |
+
"content": "The mentorship includes two courses of your choice.",
|
| 318 |
+
},
|
| 319 |
+
{"role": "user", "content": "Tell me about mentorship."},
|
| 320 |
+
],
|
| 321 |
+
"context": {"url": "https://towardsai.com/academy/mentorship/"},
|
| 322 |
+
}
|
| 323 |
+
)
|
| 324 |
+
|
| 325 |
+
retrieval_query = api._retrieval_query(payload)
|
| 326 |
+
|
| 327 |
+
assert "two courses" not in retrieval_query
|
| 328 |
+
assert "Tell me about mentorship" in retrieval_query
|
| 329 |
+
assert "Is that accurate" in retrieval_query
|
| 330 |
+
|
| 331 |
+
|
| 332 |
+
def test_api_exposes_only_sources_for_chunks_cited_by_validated_answer(
|
| 333 |
+
monkeypatch: pytest.MonkeyPatch,
|
| 334 |
+
) -> None:
|
| 335 |
+
selected = [
|
| 336 |
+
{
|
| 337 |
+
"chunk_id": "course-a:overview",
|
| 338 |
+
"title": "Course A",
|
| 339 |
+
"url": "https://towardsai.com/academy/course-a/",
|
| 340 |
+
"kind": "course",
|
| 341 |
+
"headings": ["Overview"],
|
| 342 |
+
"text": "Course A is available.",
|
| 343 |
+
},
|
| 344 |
+
{
|
| 345 |
+
"chunk_id": "course-b:pricing",
|
| 346 |
+
"title": "Course B",
|
| 347 |
+
"url": "https://towardsai.com/academy/course-b/",
|
| 348 |
+
"kind": "course",
|
| 349 |
+
"headings": ["Pricing"],
|
| 350 |
+
"text": "Course B costs $20.",
|
| 351 |
+
},
|
| 352 |
+
]
|
| 353 |
+
cited = llm.EvidenceChunk(
|
| 354 |
+
chunk_id="course-b:pricing",
|
| 355 |
+
title="Course B",
|
| 356 |
+
url="https://towardsai.com/academy/course-b/",
|
| 357 |
+
kind="course",
|
| 358 |
+
headings=("Pricing",),
|
| 359 |
+
text="Course B costs $20.",
|
| 360 |
+
)
|
| 361 |
+
grounded = llm.GroundingResult(
|
| 362 |
+
valid=True,
|
| 363 |
+
status="answered",
|
| 364 |
+
answer="Course B costs $20.",
|
| 365 |
+
claims=(
|
| 366 |
+
llm.GroundedClaim(
|
| 367 |
+
text="Course B costs $20.",
|
| 368 |
+
chunk_id="course-b:pricing",
|
| 369 |
+
quote="Course B costs $20.",
|
| 370 |
+
),
|
| 371 |
+
),
|
| 372 |
+
cited_chunks=(cited,),
|
| 373 |
+
)
|
| 374 |
+
monkeypatch.setattr(api, "page_is_allowed", lambda _url: True)
|
| 375 |
+
monkeypatch.setattr(api, "in_scope", lambda _query, _history: True)
|
| 376 |
+
monkeypatch.setattr(api, "_check_rate_limits", lambda _request, _payload: None)
|
| 377 |
+
monkeypatch.setattr(api, "_schedule_monitor", lambda _monitor: None)
|
| 378 |
+
monkeypatch.setattr(
|
| 379 |
+
api,
|
| 380 |
+
"retrieve",
|
| 381 |
+
lambda _query, *, current_url="", limit=7: selected,
|
| 382 |
+
)
|
| 383 |
+
monkeypatch.setattr(api.llm, "build_prompt", lambda **_kwargs: "prompt")
|
| 384 |
+
monkeypatch.setattr(
|
| 385 |
+
api.llm,
|
| 386 |
+
"generate_grounded_answer",
|
| 387 |
+
lambda _prompt, _selected, **_kwargs: grounded,
|
| 388 |
+
)
|
| 389 |
+
|
| 390 |
+
response = TestClient(api.app).post(
|
| 391 |
+
"/api/helper/chat",
|
| 392 |
+
headers={"Origin": "https://towardsai.com"},
|
| 393 |
+
json={
|
| 394 |
+
"query": "I want help deciding which course to take.",
|
| 395 |
+
"selectedPrompt": "I want help deciding which course to take.",
|
| 396 |
+
"visitorId": "retrieval-safety-test",
|
| 397 |
+
"history": [],
|
| 398 |
+
"context": {
|
| 399 |
+
"url": "https://towardsai.com/academy/course-b/",
|
| 400 |
+
"pageTitle": "Course B",
|
| 401 |
+
"signedIn": False,
|
| 402 |
+
},
|
| 403 |
+
},
|
| 404 |
+
)
|
| 405 |
+
|
| 406 |
+
assert response.status_code == 200
|
| 407 |
+
assert response.json()["status"] == "answered"
|
| 408 |
+
assert response.json()["sources"] == [
|
| 409 |
+
{
|
| 410 |
+
"title": "Course B",
|
| 411 |
+
"url": "https://towardsai.com/academy/course-b/",
|
| 412 |
+
"kind": "course",
|
| 413 |
+
}
|
| 414 |
+
]
|
tests/test_widget.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
from tai_helper.api import CONTACT_FORM_URL
|
| 4 |
+
from tai_helper.settings import repo_root
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
def test_widget_network_failures_keep_a_clickable_contact_handoff() -> None:
|
| 8 |
+
source = (repo_root() / "static" / "widget.js").read_text()
|
| 9 |
+
|
| 10 |
+
assert f'var contactFormUrl = "{CONTACT_FORM_URL}";' in source
|
| 11 |
+
assert "contactLinkMarkdown" in source
|
| 12 |
+
assert "The helper is unavailable on this page. Please " in source
|
| 13 |
+
assert "Something went wrong. Please " in source
|
| 14 |
+
assert "renderMarkdown(" in source
|