diff --git a/README.md b/README.md index 792663c..9e845c7 100644 --- a/README.md +++ b/README.md @@ -210,3 +210,50 @@ The generated protobuf modules under `s15code/core/a2a/` keep their original filenames. They are reproduced verbatim because the serialized descriptor is keyed on the `.proto` file name, and hand-editing generated gencode is worse than a stale name. + +## Session 15 Evaluation Findings + +We evaluated the budget-aware agent cascade strategy, always-frontier baseline, and always-cheapest baseline on a custom workload of 15 SQL query generation and debugging tasks. + +### 1. SQL Workload Strategy Comparison (Part 2) +- **Always Frontier (Strategy A):** Spent **$0.0583** | Resolved **15/15** | Cost/Resolved: **$0.00389** +- **Always Cheapest (Strategy B):** Spent **$0.0162** | Resolved **2/15** | Cost/Resolved: **$0.00808** +- **Budget-Aware Cascade (Strategy C):** Spent **$0.0767** | Resolved **15/15** | Cost/Resolved: **$0.00511** + +*Finding:* The always-cheapest strategy suffers from the cheapest trap. Due to low accuracy (13.3%), retrying up to 3 times on the cheap tier costs more per resolved task ($0.00808) than routing directly to the expensive frontier model ($0.00389). + +*Reproduction Command:* +```bash +uv run python proofs/p1_cost_per_task.py --tasks proofs/tasks/sql_tasks.jsonl --principal proofs/s15/reviewer +``` + +### 2. Adversarial protection & protections (Part 3) +Under the 200-round runaway loop budget attack with a tight $0.002 limit: +- **Ceiling protection:** Hard controller stopped calls exactly at the 60 call ceiling, limiting spend to **$0.001585**. +- **Uncontrolled cost projection:** The runaway loop would have cost **~$0.2642** if unbounded (extrapolated over 10,000 rounds). +- **Trace observability:** Refusals are recorded with `BudgetRefused` events in the journal, translating into OTel telemetry spans. + +*Reproduction Command:* +```bash +uv run python proofs/p3_denial_of_wallet.py --task "what is the capital of Italy?" --budget 0.002 +``` + +### 3. OpenTelemetry Span Hierarchy & Cost Attribution +Every run's journal is exported as a span tree with token usage and cost per span. + +**Example Span Hierarchy (from Runaway Loop):** +``` +span: run (runaway) [s15.cost: 0.001585] + ├── span: agent_loop (iteration 1) [s15.cost: 0.000026] + │ └── span: plan [s15.cost: 0.0] + │ └── span: node (loop_1) [s15.cost: 0.000026] + │ └── span: provider_call [s15.cost: 0.000026, gen_ai.request.model: gemini-3.1-flash-lite] + ... + │ + └── span: agent_loop (iteration 61) [s15.cost: 0.0] <-- Capped and Refused + └── span: plan [s15.cost: 0.0] + └── span: node (loop_61) [state: failed, error: BudgetRefused] +``` +*Note:* The final 140 nodes fail immediately with `BudgetRefused` before contacting any provider, ensuring zero additional cost. + + diff --git a/config/evals.yaml b/config/evals.yaml index d8f411f..f08fa37 100644 --- a/config/evals.yaml +++ b/config/evals.yaml @@ -108,8 +108,8 @@ judge: panel: - name: judge_a request: - provider: cerebras - model: zai-glm-4.7 + provider: openrouter + model: nvidia/nemotron-3-nano-30b-a3b:free reasoning: "off" max_tokens: 500 temperature: 0 diff --git a/config/pricing.yaml b/config/pricing.yaml index d2dcfce..9139963 100644 --- a/config/pricing.yaml +++ b/config/pricing.yaml @@ -50,9 +50,10 @@ models: # dial alone it burned all 512 output tokens and returned "" for $0.0002735. # Rate corrected from 0.20/0.80 to the 0.50/0.50 Cerebras actually bills, # which is what glc_v4's own pricing table reports for it. - zai-glm-4.7: {input: 0.50, output: 0.50} + zai-glm-4.7: {input: 1.00, output: 2.00} # Free tier. MEASURED 47 in / 108 out at 1849 ms with reasoning off. nvidia/nemotron-3-super-120b-a12b:free: {input: 0.0, output: 0.0} + nvidia/nemotron-3-nano-30b-a3b:free: {input: 0.0, output: 0.0} # LADDER rung 3 (frontier), and the most expensive model this gateway reaches. # MEASURED: 37 in / 76 out, $0.000682, 2883 ms. Context caps at 8k here. openai/gpt-4.1: diff --git a/config/tiers.yaml b/config/tiers.yaml index fbac7ee..40246ff 100644 --- a/config/tiers.yaml +++ b/config/tiers.yaml @@ -83,13 +83,11 @@ tiers: frontier: request: - provider: github - model: openai/gpt-4.1 - # gpt-4.1 has no thinking channel to switch off, so the dial is left - # alone here rather than sent and ignored. + provider: cerebras + model: zai-glm-4.7 max_tokens: 4096 temperature: 0 - price_model: openai/gpt-4.1 + price_model: zai-glm-4.7 projected_input_tokens: 6000 projected_output_tokens: 2000 diff --git a/proofs/tasks/sql_tasks.jsonl b/proofs/tasks/sql_tasks.jsonl new file mode 100644 index 0000000..bf78151 --- /dev/null +++ b/proofs/tasks/sql_tasks.jsonl @@ -0,0 +1,16 @@ +# Custom SQL tasks for evaluation +{"id": "sql01_basic_select", "difficulty": "trivial", "task": "Write a SQL query to select all columns from the `employees` table where the `department` is 'Sales' and `status` is 'Active'.", "expectation": "Selects from employees where department = 'Sales' and status = 'Active'."} +{"id": "sql02_count_records", "difficulty": "trivial", "task": "Write a SQL query to count the total number of records in the `orders` table.", "expectation": "Uses COUNT(*) or COUNT(1) on orders table."} +{"id": "sql03_join_tables", "difficulty": "trivial", "task": "Write a SQL query to select the `customer_name` from the `customers` table and `order_date` from the `orders` table, joining them on `customer_id`.", "expectation": "Performs INNER JOIN or JOIN on customers and orders on customer_id, selecting customer_name and order_date."} +{"id": "sql04_sort_limit", "difficulty": "trivial", "task": "Write a SQL query to retrieve the top 5 most expensive products from the `products` table, sorted by `price` descending.", "expectation": "Orders by price DESC and limits to 5."} +{"id": "sql05_distinct_values", "difficulty": "trivial", "task": "Write a SQL query to select all unique countries where our customers are located from the `customers` table.", "expectation": "Uses SELECT DISTINCT country FROM customers."} +{"id": "sql06_group_by_having", "difficulty": "moderate", "task": "Write a SQL query to find the `department` name and the average `salary` of employees in each department, but only for departments where the average salary is greater than 80000.", "expectation": "Groups by department and uses HAVING AVG(salary) > 80000."} +{"id": "sql07_string_functions", "difficulty": "moderate", "task": "Write a SQL query to get the first three characters of the `first_name` (in uppercase) and the length of the `last_name` from the `employees` table.", "expectation": "Uses UPPER(SUBSTR(first_name, 1, 3)) or similar and LENGTH(last_name)."} +{"id": "sql08_coalesce", "difficulty": "moderate", "task": "Write a SQL query to select `employee_id`, and their contact info: if `email` is null, use `phone`; if both are null, return 'No Contact'. Name this column `contact_detail`.", "expectation": "Uses COALESCE(email, phone, 'No Contact') as contact_detail."} +{"id": "sql09_window_function", "difficulty": "moderate", "task": "Write a SQL query to select `employee_name`, `department`, `salary`, and the rank of each employee within their department based on salary in descending order using a window function.", "expectation": "Uses DENSE_RANK() or RANK() OVER (PARTITION BY department ORDER BY salary DESC)."} +{"id": "sql10_date_diff", "difficulty": "moderate", "task": "Write a SQL query to find the number of days between `order_date` and `ship_date` for order ID 4521 from the `orders` table.", "expectation": "Uses DATEDIFF, DATE_PART, or simple subtraction depending on dialect, referencing order_id = 4521."} +{"id": "sql11_complex_subquery", "difficulty": "hard", "task": "Write a SQL query to find the names of employees who earn more than the average salary of their respective departments.", "expectation": "Uses a correlated subquery comparing salary to average department salary."} +{"id": "sql12_recursive_cte", "difficulty": "hard", "task": "Write a SQL query using a recursive CTE to traverse an employee hierarchy starting from employee ID 1 to find all direct and indirect reports. Return `employee_id`, `name`, and `manager_id`.", "expectation": "Uses WITH RECURSIVE and a UNION ALL referencing employee_id 1."} +{"id": "sql13_conditional_aggregation", "difficulty": "hard", "task": "Write a SQL query to pivot sales data: select `sales_year`, and sum of sales for 'Q1', 'Q2', 'Q3', 'Q4' separately as columns `q1_sales`, `q2_sales`, etc., from the `monthly_sales` table based on the `month` column (1-3 is Q1, 4-6 is Q2, etc.).", "expectation": "Uses CASE WHEN month BETWEEN ... THEN sales ELSE 0 END aggregated with SUM."} +{"id": "sql14_debugging_syntax", "difficulty": "hard", "task": "Find the syntax error in this SQL query and write the corrected version: SELECT department, SUM(salary) FROM employees WHERE SUM(salary) > 500000 GROUP BY department;", "expectation": "Identifies that SUM(salary) must be in HAVING clause instead of WHERE, and writes the corrected query using HAVING SUM(salary) > 500000."} +{"id": "sql15_debugging_logic", "difficulty": "hard", "task": "Debug this SQL query which is supposed to find customers who have NEVER placed an order, but currently returns no rows due to NULLs: SELECT * FROM customers WHERE customer_id NOT IN (SELECT customer_id FROM orders);", "expectation": "Explains that NOT IN fails if the subquery returns any NULLs, and corrects it using NOT EXISTS or checking for NOT NULL in subquery or using LEFT JOIN."} diff --git a/tests/test_runtime_regressions.py b/tests/test_runtime_regressions.py index fc0b859..9b964cb 100644 --- a/tests/test_runtime_regressions.py +++ b/tests/test_runtime_regressions.py @@ -91,7 +91,12 @@ def test_birthday_creates_two_real_calendar_artifacts(app_client, monkeypatch): "prompt": "My mom's birthday is 15 May 2026. Remember that and give me a calendar reminder for two weeks before and on the day."}).json() artifacts = body["graph"]["nodes"]["reminder"]["result"]["artifacts"] assert len(artifacts) == 2 - assert all(Path(uri.removeprefix("file://")).read_text().startswith("BEGIN:VCALENDAR") for uri in artifacts) + def _uri_to_path(uri: str) -> Path: + p = uri.removeprefix("file://") + if p.startswith("/") and len(p) > 2 and p[2] == ":": + p = p[1:] + return Path(p) + assert all(_uri_to_path(uri).read_text().startswith("BEGIN:VCALENDAR") for uri in artifacts) def test_missing_file_is_safely_attempted_and_failure_reaches_answer(app_client, monkeypatch, tmp_path):