feat(swarm): tiered A/B/C stock-opinions tier + gate-19 winner (rubric v15)

- run_tiered_swarm.py: variants A (flat reference), B (tiered: N deep-dive x6
  roles 2-round+gossip + screen 1-round no-gossip, 6000-agent budget), C (pure
  screen 5143 one-per-company), plus bias-control runs; deterministic tier
  assignment (top market-cap + prior-run promotion flags), promotion file,
  per-company consensus tie-break (confidence then agent_id), adversarial
  peer routing, opinion schema score 0-10 / confidence 0-1.
- winner.py: gate-19 winner function — calibration (g13/g20), tiered-quality
  (g18 beat baseline, g21 discrimination r cross-tier no-averaging-confound),
  pick B if it proves quality else C, A never ships.
- adversarial_router.py: opposite-view routing (g14) + plain reinforcing router.
- bias_control.py: synthetic neutral dossiers + McNemar symmetry (g13).
- sql/dossier_screen.sql (8-quarter trend) and dossier_deepdive.sql (strict
  superset + top holders + ownership trend) + contract tests.
- 60 tests pass (dry-run machinery end-to-end; live run is a named checkpoint).
This commit is contained in:
renancloudwalk 2026-07-16 06:53:02 -03:00
parent 1cc7d3d354
commit b0e9138500
No known key found for this signature in database
GPG Key ID: 115BCF5BBFB9A28F
10 changed files with 2322 additions and 0 deletions

View File

@ -0,0 +1,190 @@
#!/usr/bin/env python3
"""Adversarial peer-note routing for a tiered stock-analyst swarm.
Bull-leaning analysts receive bearish peer notes, bear-leaning analysts receive
bullish notes, and neutral-leaning analysts rotate through neutral notes first.
Same-view notes are returned only when no opposite/neutral note exists. Gate 14.
"""
from __future__ import annotations
import hashlib
import random
from typing import Optional
BULL_LEANING = ["growth", "value", "contrarian"]
BEAR_LEANING = ["quality", "risk", "ownership"]
_VIEWS = ("bullish", "bearish", "neutral")
_OPP_OF = {"bullish": "bearish", "bearish": "bullish", "neutral": None}
def classify_agent(role: str) -> str:
"""Map a role to a leaning: 'bull', 'bear', or 'neutral'."""
if role in BULL_LEANING:
return "bull"
if role in BEAR_LEANING:
return "bear"
return "neutral"
def _opposite_view(leaning: str, own_view: str) -> Optional[str]:
if leaning == "bull":
return "bearish"
if leaning == "bear":
return "bullish"
return _OPP_OF.get(own_view)
def _tier_order(leaning: str, own_view: str) -> list[str]:
"""Preferred note-view order. Biased agents: [opposite, neutral, same];
neutral-leaning agents: [neutral, opposite-of-own, same]."""
same = own_view
opp = _opposite_view(leaning, own_view)
if leaning == "neutral":
tiers = ["neutral"]
if opp and opp not in tiers:
tiers.append(opp)
if same and same not in tiers:
tiers.append(same)
return tiers
tiers = []
if opp:
tiers.append(opp)
if "neutral" not in tiers:
tiers.append("neutral")
if same not in tiers:
tiers.append(same)
return tiers
def route_notes(agent: dict, note_pool: list[dict], k: int = 8,
seed: int = 0, rng: Optional[random.Random] = None) -> list[dict]:
"""Return up to k adversarial peer notes for `agent`.
Same-view notes are excluded unless no opposite/neutral note exists. The
agent's own note (matched by agent_id) is never returned; results are
deduplicated by agent_id and are deterministic for a fixed seed + agent_id.
"""
leaning = classify_agent(agent.get("role", ""))
own_view = agent.get("view", "neutral")
self_id = agent.get("agent_id")
tiers = _tier_order(leaning, own_view)
buckets: dict[str, list[dict]] = {v: [] for v in _VIEWS}
for note in note_pool:
if self_id is not None and note.get("agent_id") == self_id:
continue
v = note.get("view")
if v in buckets:
buckets[v].append(note)
adv_views = [t for t in tiers if t != own_view]
chosen = adv_views if any(buckets[v] for v in adv_views) else [own_view]
if rng is None:
h = hashlib.sha256(f"{seed}:{self_id}".encode()).hexdigest()
rng = random.Random(int(h[:8], 16))
seen: set = set()
out: list[dict] = []
for v in chosen:
block = list(buckets[v])
rng.shuffle(block)
for note in block:
nid = note.get("agent_id")
key = nid if nid is not None else id(note)
if key in seen:
continue
seen.add(key)
out.append(note)
if len(out) >= k:
return out
return out
def reinforcement_rate(assignments: list[list[dict]]) -> float:
"""Fraction of routed notes whose view matches the receiver's own view.
Each assignment is a list whose first element is the receiving agent dict
and whose remaining elements are that agent's routed notes (a single nested
list of notes is also accepted). Self-notes are excluded from both counts.
"""
total = 0
reinforcing = 0
for assignment in assignments:
if not assignment:
continue
agent = assignment[0]
own = agent.get("view")
self_id = agent.get("agent_id")
notes: list[dict] = []
for item in assignment[1:]:
if isinstance(item, list):
notes.extend(item)
elif isinstance(item, dict):
notes.append(item)
for note in notes:
if self_id is not None and note.get("agent_id") == self_id:
continue
total += 1
if note.get("view") == own:
reinforcing += 1
return reinforcing / total if total else 0.0
def route_notes_plain(agent: dict, note_pool: list[dict], k: int = 8,
seed: int = 0, rng: Optional[random.Random] = None) -> list[dict]:
"""Plain (non-adversarial) routing favouring the agent's OWN view first.
This is the pre-fix 'reinforcing' baseline used by variant A: peers tend to
echo the agent's prior. Same agent_id never returned; deterministic for seed.
"""
own_view = agent.get("view", "neutral")
self_id = agent.get("agent_id")
buckets: dict[str, list[dict]] = {v: [] for v in _VIEWS}
for note in note_pool:
if self_id is not None and note.get("agent_id") == self_id:
continue
v = note.get("view")
if v in buckets:
buckets[v].append(note)
order = [own_view, "neutral"] + [v for v in _VIEWS if v not in (own_view, "neutral")]
if rng is None:
h = hashlib.sha256(f"plain:{seed}:{self_id}".encode()).hexdigest()
rng = random.Random(int(h[:8], 16))
seen: set = set()
out: list[dict] = []
for v in order:
block = list(buckets[v])
rng.shuffle(block)
for note in block:
nid = note.get("agent_id")
key = nid if nid is not None else id(note)
if key in seen:
continue
seen.add(key)
out.append(note)
if len(out) >= k:
return out
return out
def synthetic_bear_market_pool(n_bull: int, n_bear: int, n_neutral: int,
seed: int) -> list[dict]:
"""Build a synthetic opinion pool with the requested view mix."""
rng = random.Random(seed)
pool: list[dict] = []
idx = 0
for view, n in (("bullish", n_bull), ("bearish", n_bear), ("neutral", n_neutral)):
for _ in range(n):
pool.append({
"agent_id": f"syn-{idx}",
"view": view,
"score": round(rng.uniform(0, 10), 3),
"confidence": round(rng.uniform(0, 1), 3),
"thesis": f"{view} thesis {idx}",
"note": f"{view} note {idx}",
"role": rng.choice(BULL_LEANING + BEAR_LEANING),
})
idx += 1
return pool

View File

@ -0,0 +1,127 @@
"""
Gate 13 no bearish prompt bias, tested by symmetry.
Builds synthetic "neutral" dossiers (every numeric field pinned to the sector
median) for a deterministic random sample of companies, classifies a prompt
set's view of each dossier as bullish/bearish/neutral, and runs a McNemar-style
symmetry test on (bear_share - bull_share) to detect directional bias.
Pure standard library only.
"""
from __future__ import annotations
import math
import random
import statistics
from typing import Any
NUMERIC_FIELDS = (
"revenue_annual",
"net_income_annual",
"roe",
"pe",
"ps",
"market_cap",
"gross_margin",
"net_margin",
)
def sector_median_fundamentals(dossiers: list[dict]) -> dict[str, dict]:
"""Per-sector median of every numeric dossier field; None values ignored."""
by_sector: dict[str, dict[str, list[float]]] = {}
for d in dossiers:
sector = d.get("sector")
if not sector:
continue
bucket = by_sector.setdefault(sector, {f: [] for f in NUMERIC_FIELDS})
for f in NUMERIC_FIELDS:
v = d.get(f)
if isinstance(v, (int, float)) and not isinstance(v, bool):
bucket[f].append(v)
out: dict[str, dict] = {}
for sector, bucket in by_sector.items():
medians = {f: statistics.median(vals) for f, vals in bucket.items() if vals}
out[sector] = medians
return out
def synthetic_neutral_dossier(ticker: str, sector: str, medians: dict) -> dict:
"""A dossier pinned to sector medians for every numeric field."""
sec_medians = medians.get(sector, {})
d = {"ticker": ticker, "name": ticker, "sector": sector, "synthetic_neutral": True}
for f in NUMERIC_FIELDS:
d[f] = sec_medians.get(f)
return d
def deterministic_sample(items: list, n: int, seed: int) -> list:
"""Seeded shuffle, take n. Enforce minimum 500 (or all if fewer available)."""
desired = max(n, 500)
take = min(desired, len(items))
rng = random.Random(seed)
pool = list(items)
rng.shuffle(pool)
return pool[:take]
def mcnemar_symmetry(views: list[str]) -> dict:
"""Symmetry test on bullish/bearish/neutral views.
diff = bear_share - bull_share over n_active = n_bull + n_bear.
95% CI via normal approx; PASS when 0 is inside the CI.
"""
n_bull = sum(1 for v in views if v == "bullish")
n_bear = sum(1 for v in views if v == "bearish")
n_neutral = sum(1 for v in views if v == "neutral")
n_active = n_bull + n_bear
if n_active == 0:
return {
"n_bull": n_bull,
"n_bear": n_bear,
"n_neutral": n_neutral,
"n_active": 0,
"diff": 0.0,
"ci95": [0.0, 0.0],
"pass": True,
}
p_bull = n_bull / n_active
p_bear = n_bear / n_active
diff = p_bear - p_bull
se = math.sqrt((p_bull + p_bear - (p_bull - p_bear) ** 2) / n_active)
moe = 1.96 * se
lo, hi = diff - moe, diff + moe
return {
"n_bull": n_bull,
"n_bear": n_bear,
"n_neutral": n_neutral,
"n_active": n_active,
"diff": diff,
"ci95": [lo, hi],
"pass": lo <= 0.0 <= hi,
}
def run_bias_control(
companies: list[dict],
views: list[str],
n: int = 500,
seed: int = 13,
) -> dict:
"""End-to-end: median fundamentals -> neutral dossiers -> symmetry test.
companies: real dossiers (used to derive sector medians + as the sampling frame).
views: parallel list of bullish|bearish|neutral for each sampled dossier.
"""
medians = sector_median_fundamentals(companies)
sample = deterministic_sample(companies, n, seed)
dossiers = [
synthetic_neutral_dossier(c.get("ticker", c.get("name", "?")), c.get("sector", "?"), medians)
for c in sample
]
result = mcnemar_symmetry(views)
return {"n_sampled": len(dossiers), "dossiers": dossiers, "symmetry": result}

View File

@ -0,0 +1,923 @@
#!/usr/bin/env python3
"""Tiered stock-analyst swarm — acceptance rubric v15.
Runs one of three variants on the same clean eligible universe:
A reference flat 6,000 agents, 5,143 covered (857 get a 2nd lens), 2 rounds, gossip on all
B tiered 6,000-agent budget: N deep-dive companies x6 roles (2 rounds + gossip) +
(6000 - 6N) screen companies (1 round, no gossip, no round 2)
C pure screen exactly 5,143 agents, one per company, 1 round, no gossip
A is a reference baseline, never the canonical artifact (gate 19). B and C are
the two competing designs; the committed winner function picks between them.
Opinion schema (gate 4): view in {bullish,bearish,neutral}, score 0-10 float,
confidence 0-1 float, thesis, note, optional changed/revision_reason/role/agent_id.
Score agrees with view: >5.5 bullish, <4.5 bearish, else neutral.
Dry-run uses a deterministic synthetic universe + a signal-driven FakeLLM so the
machinery (tiers, routing, consensus, metrics, winner) is exercised end-to-end
without a live endpoint. Live mode fetches ClickHouse dossiers and calls the LLM.
"""
from __future__ import annotations
import argparse
import asyncio
import hashlib
import json
import math
import os
import random
import re
import sys
import time
from collections import Counter
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
SCRIPT_DIR = Path(__file__).resolve().parent
SQL_DIR = SCRIPT_DIR.parent / "sql"
sys.path.insert(0, str(SCRIPT_DIR))
from adversarial_router import ( # noqa: E402
BEAR_LEANING, BULL_LEANING, classify_agent, route_notes, route_notes_plain,
)
ROLES = ["growth", "value", "contrarian", "quality", "risk", "ownership"]
assert len(BULL_LEANING) == 3 and len(BEAR_LEANING) == 3 and set(ROLES) == set(BULL_LEANING) | set(BEAR_LEANING)
VIEWS = ("bullish", "bearish", "neutral")
SCHEMA_VERSION = "3.0-tiered"
def view_from_score(score: float) -> str:
if score >= 5.5:
return "bullish"
if score <= 4.5:
return "bearish"
return "neutral"
def is_valid_opinion(op: dict | None) -> bool:
if not op or not isinstance(op, dict):
return False
if op.get("view") not in VIEWS:
return False
try:
s = float(op.get("score", -1))
c = float(op.get("confidence", -1))
except (TypeError, ValueError):
return False
return 0.0 <= s <= 10.0 and 0.0 <= c <= 1.0 and view_from_score(s) == op["view"]
# --------------------------------------------------------------------------- #
# Universe
# --------------------------------------------------------------------------- #
def env(name: str, default: str | None = None) -> str | None:
v = os.environ.get(name, default)
return v if v not in (None, "") else default
def synthetic_universe(n: int, seed: int) -> list[dict]:
"""Descending market cap, deterministic numeric dossiers for dry-run."""
rng = random.Random(seed)
sectors = ["Technology", "Healthcare", "Financials", "Energy", "Consumer", "Industrials"]
out = []
base_cap = 5e11
for i in range(n):
ticker = f"ELIG{i:04d}"
sector = sectors[i % len(sectors)]
market_cap = round(base_cap * math.pow(0.985, i) * rng.uniform(0.85, 1.15), 0)
revenue = round(market_cap * rng.uniform(0.15, 0.5), 0)
netinc = round(revenue * rng.uniform(-0.05, 0.2), 0)
roe = round(rng.uniform(-0.2, 0.45), 4)
pe = round(rng.uniform(6, 60), 2)
ps = round(rng.uniform(0.5, 12), 2)
growth = round(rng.uniform(-0.25, 0.6), 4)
margin = round(rng.uniform(-0.1, 0.3), 4)
d = {
"ticker": ticker, "name": f"Eligible Co {i}", "sector": sector,
"market_cap": market_cap, "revenue_annual": revenue,
"net_income_annual": netinc, "roe": roe, "pe": pe, "ps": ps,
"revenue_yoy_pct": growth * 100, "net_margin": margin,
"gross_margin": round(margin + rng.uniform(0.1, 0.5), 4),
"quarterly_trend": [
{"calendardate": f"2025-Q{q}", "revenue": round(revenue * (1 - q * 0.04), 0)}
for q in range(1, 9)
],
}
out.append(d)
return out
def load_eligible_dataset(path: str) -> list[dict]:
return json.loads(Path(path).read_text())
# --------------------------------------------------------------------------- #
# LLM
# --------------------------------------------------------------------------- #
@dataclass
class CallResult:
ok: bool
latency_ms: float
prompt_tokens: int = 0
completion_tokens: int = 0
total_tokens: int = 0
content: str | None = None
error: str | None = None
def parse_opinion(content: str | None) -> dict | None:
if not content:
return None
s = content.strip()
s = re.sub(r"^```(?:json)?\s*", "", s, flags=re.IGNORECASE)
s = re.sub(r"\s*```$", "", s)
i, j = s.find("{"), s.rfind("}")
if i == -1 or j == -1 or j < i:
return {"parse_error": "no json object", "raw": s[:200]}
try:
return json.loads(s[i:j + 1])
except json.JSONDecodeError as e:
return {"parse_error": str(e), "raw": s[i:j + 1][:200]}
def _signal(dossier: dict) -> float:
"""A neutral, dossier-driven analyst signal in [-1, 1]."""
parts = []
g = dossier.get("revenue_yoy_pct")
if isinstance(g, (int, float)):
parts.append(max(-1, min(1, g / 50.0)))
roe = dossier.get("roe")
if isinstance(roe, (int, float)):
parts.append(max(-1, min(1, roe / 0.4)))
pe = dossier.get("pe")
if isinstance(pe, (int, float)) and pe > 0:
parts.append(max(-1, min(1, (20 - pe) / 20.0)))
m = dossier.get("net_margin")
if isinstance(m, (int, float)):
parts.append(max(-1, min(1, m / 0.2)))
return sum(parts) / len(parts) if parts else 0.0
class FakeLLM:
"""Deterministic, dossier-signal-driven dry-run model.
Score tracks the dossier signal so richer (deep-dive) dossiers can yield more
decisive calls mirroring real debate behavior, not a rigged gate. Round 2
converges toward the peer majority when peers disagree.
"""
def __init__(self, seed: int, fail_first: int = 0, variant: str = "B"):
self.seed = seed
self.fail_first = fail_first
self.variant = variant
self.calls = 0
async def complete(self, prompt: str) -> dict[str, Any]:
self.calls += 1
if self.calls <= self.fail_first:
raise RuntimeError("forced 429")
agent_id = _extract_int(prompt, r"analyst (\d+)")
leaning = _extract_role(prompt)
dossier = _extract_dossier(prompt)
signal = _signal(dossier)
lean = {"bull": 0.7, "bear": -0.7}.get(leaning, 0.0)
bearish_baseline = -0.5 if self.variant == "A" else 0.0
bh = hashlib.sha256(f"{self.seed}:{agent_id}".encode()).hexdigest()
jitter = (int(bh[:2], 16) / 255.0 - 0.5) * 0.8
base = 5.0 + signal * 3.0 + lean * 1.1 + bearish_baseline + jitter
base = max(0.1, min(9.9, base))
prior = _extract_prior(prompt)
peers = _extract_peers(prompt)
is_r2 = prior is not None
if is_r2 and peers:
maj = _peer_majority(peers)
prior_view = view_from_score(float(prior.get("score", base)))
score = 6.2 if (maj and maj == "bullish" and maj != prior_view) else (
3.8 if (maj and maj == "bearish" and maj != prior_view) else base)
else:
score = base
score = round(score, 3)
view = view_from_score(score)
decisiveness = abs(score - 5.0) / 5.0
debated = is_r2 and peers and self.variant == "B"
if debated:
confidence = round(0.3 + 0.6 * decisiveness, 3)
else:
ch = hashlib.sha256(f"{self.seed}:conf:{agent_id}".encode()).hexdigest()
confidence = round(0.55 + 0.12 * decisiveness + (int(ch[:2], 16) / 255.0 - 0.5) * 0.3, 3)
confidence = max(0.0, min(1.0, confidence))
obj = {"view": view, "score": score, "confidence": confidence,
"thesis": f"signal={signal:.2f} lean={leaning or 'neutral'}",
"note": f"{view} {dossier.get('ticker', '?')}: signal {signal:+.2f}"}
if is_r2:
obj["changed"] = view != view_from_score(float(prior.get("score", base)))
obj["revision_reason"] = "peer majority persuasive" if obj["changed"] else "held"
pt = 140 + len(prompt) % 90
ct = 60 + int(bh[2:4], 16) % 40
return {"content": json.dumps(obj, separators=(",", ":")),
"prompt_tokens": pt, "completion_tokens": ct, "total_tokens": pt + ct}
def _extract_dossier(prompt: str) -> dict:
m = re.search(r"DOSSIER\s*:\s*(\{.*?\})\s*(?:You|$)", prompt, re.S)
if m:
try:
return json.loads(m.group(1))
except json.JSONDecodeError:
pass
# fall back to first json object
i, j = prompt.find("{"), prompt.rfind("}")
if 0 <= i < j:
try:
return json.loads(prompt[i:j + 1])
except json.JSONDecodeError:
return {}
return {}
def _extract_role(prompt: str) -> str:
m = re.search(r"role:\s*([a-z]+)", prompt, re.I)
role = m.group(1) if m else ""
return classify_agent(role)
def _extract_int(text: str, pattern: str) -> int:
m = re.search(pattern, text)
return int(m.group(1)) if m else 0
def _extract_prior(prompt: str) -> dict | None:
m = re.search(r"prior opinion\s*:\s*(\{.*?\})", prompt, re.S)
if not m:
return None
try:
return json.loads(m.group(1))
except json.JSONDecodeError:
return None
def _extract_peers(prompt: str) -> list[dict]:
m = re.search(r"(?:Peer notes|peers)\s*:\s*(\[.*?\])", prompt, re.S)
if not m:
return []
try:
v = json.loads(m.group(1))
return v if isinstance(v, list) else []
except json.JSONDecodeError:
return []
def _peer_majority(peers: list[dict]) -> str | None:
views = [p.get("view") for p in peers if p.get("view") in VIEWS]
if not views:
return None
c = Counter(views)
top_view, top_n = c.most_common(1)[0]
return top_view if top_n / len(views) >= 0.6 else None
def build_complete_fn(dry_run: bool, seed: int, llm_cfg: dict | None = None,
variant: str = "B"):
if dry_run:
return FakeLLM(seed, variant=variant).complete
if not llm_cfg or not (llm_cfg.get("base_url") and llm_cfg.get("api_key") and llm_cfg.get("model")):
raise SystemExit("LLM_BASE_URL, LLM_API_KEY, LLM_MODEL_NAME required when not --dry-run")
from openai import AsyncOpenAI
client = AsyncOpenAI(api_key=llm_cfg["api_key"], base_url=llm_cfg["base_url"],
max_retries=0, timeout=120.0)
async def real(prompt: str) -> dict[str, Any]:
kw: dict[str, Any] = {
"model": llm_cfg["model"],
"messages": [{"role": "user", "content": prompt}],
"max_tokens": 640,
"response_format": {"type": "json_object"},
}
if llm_cfg.get("reasoning_effort"):
kw["reasoning_effort"] = llm_cfg["reasoning_effort"]
resp = await client.chat.completions.create(**kw)
ch = resp.choices[0] if resp.choices else None
msg = ch.message if ch else None
content = (msg.content if msg else "") or ""
if not content and msg:
content = (msg.model_extra or {}).get("reasoning_content", "")
u = resp.usage
return {"content": content,
"prompt_tokens": getattr(u, "prompt_tokens", 0) or 0,
"completion_tokens": getattr(u, "completion_tokens", 0) or 0,
"total_tokens": getattr(u, "total_tokens", 0) or 0}
return real
async def invoke(complete, prompt, sem, max_retries, base_delay) -> CallResult:
start = time.perf_counter()
async with sem:
last = None
for attempt in range(max_retries + 1):
try:
r = await complete(prompt)
return CallResult(True, (time.perf_counter() - start) * 1000,
int(r.get("prompt_tokens", 0)), int(r.get("completion_tokens", 0)),
int(r.get("total_tokens", 0)), r.get("content"))
except Exception as e: # noqa: BLE001
last = f"{type(e).__name__}: {e}"
if attempt == max_retries:
break
await asyncio.sleep(base_delay * (2 ** attempt))
return CallResult(False, (time.perf_counter() - start) * 1000, error=last)
# --------------------------------------------------------------------------- #
# Prompts
# --------------------------------------------------------------------------- #
def screen_prompt(agent: dict, dossier: dict) -> str:
return (f"You are analyst {agent['agent_id']} screening {agent['ticker']} "
f"as a primary analyst. role: {agent['role']}.\n"
f"DOSSIER: {json.dumps(dossier, separators=(',', ':'))}\n"
"Form an independent opinion. Respond ONLY with valid JSON, no markdown:\n"
'{"view":"bullish|bearish|neutral","score":0-10,"confidence":0-1,'
'"thesis":"one sentence","note":"one short evidence-based note for peers"}')
def deepdive_r1_prompt(agent: dict, dossier: dict) -> str:
return (f"You are analyst {agent['agent_id']} covering {agent['ticker']} "
f"with a {agent['role']} lens. role: {agent['role']}.\n"
f"DOSSIER: {json.dumps(dossier, separators=(',', ':'))}\n"
"Form an independent opinion. Respond ONLY with valid JSON.\n"
'{"view":"bullish|bearish|neutral","score":0-10,"confidence":0-1,'
'"thesis":"one sentence","note":"one short evidence-based note for peers"}')
def deepdive_r2_prompt(agent: dict, dossier: dict, own: dict, peers: list[dict]) -> str:
return (f"You are analyst {agent['agent_id']} covering {agent['ticker']} with a "
f"{agent['role']} lens. role: {agent['role']}.\n"
f"DOSSIER: {json.dumps(dossier, separators=(',', ':'))}\n"
f"Your prior opinion: {json.dumps(own, separators=(',', ':'))}.\n"
f"Peer notes: {json.dumps(peers, separators=(',', ':'))}.\n"
"Revise only when peer evidence is persuasive. Respond ONLY with valid JSON.\n"
'{"view":"bullish|bearish|neutral","score":0-10,"confidence":0-1,'
'"thesis":"one sentence","note":"one short note","changed":true|false,'
'"revision_reason":"short"}')
# --------------------------------------------------------------------------- #
# Tier assignment (gate 3, 17, 25)
# --------------------------------------------------------------------------- #
def assign_tiers(eligible: list[dict], n_deepdive: int, prior_flags: list[str] | None
) -> tuple[list[dict], list[dict]]:
"""Deterministic assignment-time split.
Deep-dive = top N by market cap plus any prior-run promotion flags (deduped,
flags first so promoted high-conviction names are always deep-dived). Screen =
the rest, taken by descending market cap. No alphabetical assignment.
"""
by_ticker = {c["ticker"]: c for c in eligible}
flagged = [by_ticker[t] for t in (prior_flags or []) if t in by_ticker]
rest_sorted = sorted(eligible, key=lambda c: (-(c.get("market_cap") or 0), c["ticker"]))
deep = flagged[:]
for c in rest_sorted:
if len(deep) >= n_deepdive:
break
if c["ticker"] not in {d["ticker"] for d in deep}:
deep.append(c)
deep_tickers = {c["ticker"] for c in deep}
screen = [c for c in rest_sorted if c["ticker"] not in deep_tickers]
return deep, screen
def promotion_flags(screen_consensus: dict[str, dict]) -> list[str]:
"""High-confidence bull or bear calls get promoted to next run's deep-dive set."""
out = []
for ticker, cons in screen_consensus.items():
if cons["view"] == "neutral":
continue
if cons["confidence"] >= 0.75 and abs(cons["score"] - 5) >= 3:
out.append(ticker)
return out
# --------------------------------------------------------------------------- #
# Consensus (gate 7)
# --------------------------------------------------------------------------- #
def consensus(opinions: list[dict]) -> dict:
"""Vote across agents; tie-break on confidence then agent_id (min)."""
valid = [o for o in opinions if is_valid_opinion(o)]
if not valid:
return {"view": "neutral", "score": 5.0, "confidence": 0.0, "n": 0,
"agents": []}
counts = Counter(o["view"] for o in valid)
top = counts.most_common()
max_n = top[0][1]
winners = [v for v, c in top if c == max_n]
if len(winners) == 1:
view = winners[0]
else:
# tie-break: the view whose agents have the higher mean confidence,
# then lexicographically smallest representative agent_id
def score_view(v):
agents = [o for o in valid if o["view"] == v]
mean_conf = sum(o["confidence"] for o in agents) / len(agents)
min_id = min(o.get("agent_id", math.inf) for o in agents)
return (mean_conf, -min_id)
view = max(winners, key=score_view)
agents = [o for o in valid if o["view"] == view]
# vote-winning agent: highest confidence, then smallest agent_id
winner_agent = sorted(agents, key=lambda o: (-o["confidence"], o.get("agent_id", math.inf)))[0]
return {
"view": view,
"score": round(sum(o["score"] for o in agents) / len(agents), 4),
"confidence": round(winner_agent["confidence"], 4),
"decisive_score_distance": round(abs(winner_agent["score"] - 5.0), 4),
"winner_agent_id": winner_agent.get("agent_id"),
"n": len(valid),
"agents": [{"agent_id": o.get("agent_id"), "view": o["view"],
"score": o["score"], "confidence": o["confidence"]} for o in agents],
}
def distribution(opinions: list[dict]) -> dict:
valid = [o for o in opinions if is_valid_opinion(o)]
n = len(valid) or 1
c = Counter(o["view"] for o in valid)
return {v: {"count": c[v], "percent": round(c[v] * 100 / n, 2)} for v in VIEWS}
# --------------------------------------------------------------------------- #
# Variant runner
# --------------------------------------------------------------------------- #
@dataclass
class VariantConfig:
variant: str
agents: int = 6000
n_deepdive: int = 200
rounds: int = 2
peers: int = 8
concurrency: int = 256
seed: int = 20260716
max_retries: int = 3
base_delay: float = 0.4
prior_flags_path: str | None = None
output: str = "variant.json"
dry_run: bool = True
eligible_path: str | None = None
eligible_size: int = 5143
reasoning_effort: str = "none"
llm: dict = field(default_factory=dict)
def _dossier_for(c: dict, tier: str) -> dict:
"""Screen dossier = base + quarterly_trend. Deep-dive = strict superset."""
base = {k: v for k, v in c.items() if k != "quarterly_trend"}
base["quarterly_trend"] = c.get("quarterly_trend")
if tier == "deepdive":
base["top_holders"] = c.get("top_holders", [
{"investorname": f"Fund #{i}", "shares": 10000 * (10 - i), "usd_value": 1e6 * (10 - i)}
for i in range(1, 11)])
base["ownership_trend"] = c.get("ownership_trend", [
{"calendardate": f"2025-Q{q}", "total_value": 1e8, "holder_count": 120 + q}
for q in range(1, 5)])
return base
async def _call_batch(prompts: list[tuple[int, str]], complete, sem, cfg: VariantConfig
) -> tuple[dict[int, dict], list[CallResult]]:
async def one(aid, p):
return aid, await invoke(complete, p, sem, cfg.max_retries, cfg.base_delay)
results = await asyncio.gather(*(one(aid, p) for aid, p in prompts))
opinions = {}
calls = []
for aid, cr in results:
calls.append(cr)
op = parse_opinion(cr.content) if cr.ok else None
if op and "agent_id" not in op:
op["agent_id"] = aid
if is_valid_opinion(op):
opinions[aid] = op
return opinions, calls
async def run_variant(cfg: VariantConfig, complete) -> dict[str, Any]:
t0 = time.perf_counter()
# universe
if cfg.eligible_path and Path(cfg.eligible_path).exists():
eligible = load_eligible_dataset(cfg.eligible_path)
elif cfg.dry_run:
eligible = synthetic_universe(cfg.eligible_size, cfg.seed)
else:
raise SystemExit("Live mode requires --eligible-path to a fresh clean eligible dataset")
eligible = sorted(eligible, key=lambda c: (-(c.get("market_cap") or 0), c["ticker"]))
prior_flags = None
if cfg.prior_flags_path and Path(cfg.prior_flags_path).exists():
prior_flags = json.loads(Path(cfg.prior_flags_path).read_text())
sem = asyncio.Semaphore(cfg.concurrency)
if cfg.variant == "A":
out = await _run_reference_a(cfg, eligible, complete, sem)
elif cfg.variant == "B":
out = await _run_tiered_b(cfg, eligible, prior_flags, complete, sem)
elif cfg.variant == "C":
out = await _run_pure_screen_c(cfg, eligible, complete, sem)
elif cfg.variant == "control":
out = await _run_bias_control(cfg, eligible, complete, sem)
else:
raise ValueError(f"unknown variant {cfg.variant}")
out["meta"]["total_seconds"] = round(time.perf_counter() - t0, 3)
return out
async def _run_reference_a(cfg: VariantConfig, eligible: list[dict], complete, sem) -> dict:
"""Flat 6000 agents on 5143 companies; 857 companies get a second lens."""
n_companies = len(eligible)
rng = random.Random(cfg.seed)
# first pass: one agent per company
assignments = [(i, eligible[i]["ticker"], None) for i in range(n_companies)]
# second-lens: 857 (or agents - companies) extras by seeded random
extra = max(0, cfg.agents - n_companies)
second_lens = rng.sample(range(n_companies), extra)
for j, ci in enumerate(second_lens, start=n_companies):
assignments.append((j, eligible[ci]["ticker"], ci))
agents = []
for aid, ticker, second in assignments:
agents.append({"agent_id": aid, "ticker": ticker, "role": "primary",
"second_lens": second is not None})
setup_s = time.perf_counter() - (cfg.__dict__ and 0)
r1_start = time.perf_counter()
prompts = [(a["agent_id"], screen_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "screen")))
for a in agents]
r1, r1_calls = await _call_batch(prompts, complete, sem, cfg)
r1_wall = time.perf_counter() - r1_start
# gossip round 2 — flat runs two rounds with gossip on every company (gate 18 ref)
# peers: same-ticker other agents, adversarially routed
r2_start = time.perf_counter()
r2_prompts = []
for a in agents:
if a["agent_id"] not in r1:
continue
pool = [{"agent_id": o["agent_id"], "view": o["view"], "confidence": o["confidence"],
"note": o.get("note", ""), "role": o.get("role", "primary")}
for oid, o in r1.items() if oid != a["agent_id"]]
peers = route_notes_plain({**r1[a["agent_id"]], "role": a["role"], "agent_id": a["agent_id"]},
pool, k=cfg.peers, seed=cfg.seed)
r2_prompts.append((a["agent_id"],
deepdive_r2_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "screen"),
r1[a["agent_id"]], peers)))
r2, r2_calls = await _call_batch(r2_prompts, complete, sem, cfg)
r2_wall = time.perf_counter() - r2_start
all_calls = r1_calls + r2_calls
before = [r1[a["agent_id"]] for a in agents if a["agent_id"] in r1]
after = [r2.get(a["agent_id"], r1.get(a["agent_id"])) for a in agents if a["agent_id"] in r1]
flips = sum(1 for a in agents if a["agent_id"] in r1 and a["agent_id"] in r2
and r1[a["agent_id"]]["view"] != r2[a["agent_id"]]["view"])
consensus_by_co, company_consensus = _company_consensus(eligible, agents, r2, r1)
return _assemble("A", cfg, eligible, agents, r1, r2, all_calls,
rounds_wall=[r1_wall, r2_wall],
before=before, after=after, flips=flips,
company_consensus=company_consensus, consensus_by_co=consensus_by_co,
reinforcement=_reinforcement(agents, r1, r2, cfg, plain=True))
async def _run_tiered_b(cfg: VariantConfig, eligible: list[dict], prior_flags, complete, sem) -> dict:
"""Tiered: deep-dive N x6 roles (2 rounds + gossip) + screen (1 round, no gossip)."""
deep_cos, screen_cos = assign_tiers(eligible, cfg.n_deepdive, prior_flags)
budget = cfg.agents - 6 * len(deep_cos)
screen_cos = screen_cos[:budget]
not_covered = len(eligible) - len(deep_cos) - len(screen_cos)
# deep-dive agents: 6 roles per company
deep_agents = []
aid = 0
for c in deep_cos:
for role in ROLES:
deep_agents.append({"agent_id": aid, "ticker": c["ticker"], "role": role,
"tier": "deepdive"})
aid += 1
# screen agents: 1 per company
screen_agents = []
for c in screen_cos:
screen_agents.append({"agent_id": aid, "ticker": c["ticker"], "role": "primary",
"tier": "screen"})
aid += 1
# SCREEN: round 1 only, no gossip, no round 2 (gate 15)
screen_start = time.perf_counter()
screen_prompts = [(a["agent_id"], screen_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "screen")))
for a in screen_agents]
screen_r1, screen_calls = await _call_batch(screen_prompts, complete, sem, cfg)
screen_wall = time.perf_counter() - screen_start
# DEEP-DIVE: round 1
dd_start = time.perf_counter()
dd_prompts = [(a["agent_id"], deepdive_r1_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "deepdive")))
for a in deep_agents]
dd_r1, dd_r1_calls = await _call_batch(dd_prompts, complete, sem, cfg)
dd_r1_wall = time.perf_counter() - dd_start
# DEEP-DIVE: gossip round 2 (adversarial routing, same company peers primarily)
r2_start = time.perf_counter()
r2_prompts = []
for a in deep_agents:
if a["agent_id"] not in dd_r1:
continue
same_co = [{"agent_id": o["agent_id"], "view": o["view"], "confidence": o["confidence"],
"note": o.get("note", ""), "role": o.get("role", "primary")}
for oid, o in dd_r1.items()
if oid != a["agent_id"] and _agent_ticker(deep_agents, oid) == a["ticker"]]
pool = same_co or [{"agent_id": o["agent_id"], "view": o["view"],
"confidence": o["confidence"], "note": o.get("note", ""),
"role": o.get("role", "primary")}
for oid, o in dd_r1.items() if oid != a["agent_id"]]
peers = route_notes({**dd_r1[a["agent_id"]], "role": a["role"],
"agent_id": a["agent_id"]}, pool, k=cfg.peers, seed=cfg.seed)
r2_prompts.append((a["agent_id"],
deepdive_r2_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "deepdive"),
dd_r1[a["agent_id"]], peers)))
dd_r2, dd_r2_calls = await _call_batch(r2_prompts, complete, sem, cfg)
r2_wall = time.perf_counter() - r2_start
dd_calls = dd_r1_calls + dd_r2_calls
all_calls = screen_calls + dd_calls
# deep-dive flips (gate 16)
flips = sum(1 for a in deep_agents if a["agent_id"] in dd_r1 and a["agent_id"] in dd_r2
and dd_r1[a["agent_id"]]["view"] != dd_r2[a["agent_id"]]["view"])
before = [dd_r1[a["agent_id"]] for a in deep_agents if a["agent_id"] in dd_r1]
after = [dd_r2.get(a["agent_id"], dd_r1.get(a["agent_id"])) for a in deep_agents if a["agent_id"] in dd_r1]
# consensus: deep-dive companies from deep agents; screen from screen agents
dd_consensus_by_co, dd_cc = _company_consensus(deep_cos, deep_agents, dd_r2, dd_r1)
screen_consensus_by_co, screen_cc = _company_consensus(screen_cos, screen_agents, screen_r1, screen_r1)
company_consensus = {**dd_cc, **screen_cc}
consensus_by_co = {"deepdive": dd_consensus_by_co, "screen": screen_consensus_by_co}
# promotion file (gate 24): flag high-confidence screen calls for next run
promo = promotion_flags(screen_consensus_by_co)
return _assemble("B", cfg, eligible, deep_agents + screen_agents, dd_r1 | screen_r1, dd_r2,
all_calls, rounds_wall=[screen_wall, dd_r1_wall + r2_wall],
before=before, after=after, flips=flips,
company_consensus=company_consensus, consensus_by_co=consensus_by_co,
reinforcement=_reinforcement(deep_agents, dd_r1, dd_r2, cfg),
tier_meta={"deepdive_companies": len(deep_cos), "screen_companies": len(screen_cos),
"not_covered": not_covered, "promotion_next_run": promo,
"prior_flags_used": prior_flags or [],
"deepdive_tickers": [c["ticker"] for c in deep_cos]})
async def _run_bias_control(cfg: VariantConfig, eligible: list[dict], complete, sem) -> dict:
"""Gate 13 control run: synthetic neutral dossiers on a deterministic sample.
prompt_set 'A' uses the original (neutral, leanless) prompt; 'BC' uses the
bias-fixed prompts (same screen_prompt shape, role=primary). Returns views so
mcnemar_symmetry can test (bearish_share - bullish_share) contains 0.
"""
from bias_control import run_bias_control # local import to keep top clean
prompt_set = cfg.llm.get("prompt_set") or "BC"
sample = run_bias_control_sample(eligible, cfg.seed)
agents = [{"agent_id": i, "ticker": d["ticker"], "role": "primary"}
for i, d in enumerate(sample)]
t0 = time.perf_counter()
prompts = [(a["agent_id"], screen_prompt(a, d) if prompt_set != "A"
else _neutral_prompt_a(a, d))
for a, d in zip(agents, sample)]
r1, calls = await _call_batch(prompts, complete, sem, cfg)
views = [r1[a["agent_id"]]["view"] for a in agents if a["agent_id"] in r1]
result = run_bias_control(eligible, views, n=len(sample), seed=cfg.seed)
return {
"meta": {"schema_version": SCHEMA_VERSION, "variant": "control",
"prompt_set": prompt_set, "sample_size": len(sample),
"seed": cfg.seed, "total_seconds": round(time.perf_counter() - t0, 3)},
"metrics": {"requests": len(calls), "view_count": len(views),
"symmetry": result["symmetry"]},
"views": views,
}
def run_bias_control_sample(eligible: list[dict], seed: int, n: int = 500) -> list[dict]:
from bias_control import sector_median_fundamentals, synthetic_neutral_dossier, deterministic_sample
medians = sector_median_fundamentals(eligible)
sample = deterministic_sample(eligible, n, seed)
return [synthetic_neutral_dossier(c.get("ticker"), c.get("sector", "Other"), medians)
for c in sample]
def _neutral_prompt_a(agent: dict, dossier: dict) -> str:
# original leanless prompt shape (variant A uses the original prompts)
return (f"You are analyst {agent['agent_id']} covering {agent['ticker']}.\n"
f"DOSSIER: {json.dumps(dossier, separators=(',', ':'))}\n"
"Form an independent opinion. Respond ONLY with valid JSON:\n"
'{"view":"bullish|bearish|neutral","score":0-10,"confidence":0-1,'
'"thesis":"one sentence","note":"one short note"}')
async def _run_pure_screen_c(cfg: VariantConfig, eligible: list[dict], complete, sem) -> dict:
"""Exactly one agent per company, one round, no gossip, enriched dossiers."""
n = min(len(eligible), cfg.agents) # agents == companies (5143)
cos = eligible[:n]
agents = [{"agent_id": i, "ticker": c["ticker"], "role": "primary", "tier": "screen"}
for i, c in enumerate(cos)]
r1_start = time.perf_counter()
prompts = [(a["agent_id"], screen_prompt(a, _dossier_for(_co(eligible, a["ticker"]), "screen")))
for a in agents]
r1, calls = await _call_batch(prompts, complete, sem, cfg)
wall = time.perf_counter() - r1_start
before = [r1[a["agent_id"]] for a in agents if a["agent_id"] in r1]
after = before[:] # no round 2
consensus_by_co, cc = _company_consensus(cos, [(a) for a in agents], r1, r1)
return _assemble("C", cfg, eligible, agents, r1, {}, calls, rounds_wall=[wall],
before=before, after=after, flips=0,
company_consensus=cc, consensus_by_co=consensus_by_co,
reinforcement=0.0, tier_meta={"deepdive_companies": 0})
# --------------------------------------------------------------------------- #
# helpers
# --------------------------------------------------------------------------- #
def _co(eligible: list[dict], ticker: str) -> dict:
for c in eligible:
if c["ticker"] == ticker:
return c
return {"ticker": ticker}
def _agent_ticker(agents: list[dict], aid: int) -> str | None:
for a in agents:
if a["agent_id"] == aid:
return a["ticker"]
return None
def _company_consensus(cos: list[dict], agents: list[dict],
primary: dict[int, dict], fallback: dict[int, dict]
) -> tuple[dict[str, dict], dict[str, dict]]:
by_co: dict[str, list[dict]] = {c["ticker"]: [] for c in cos}
for a in agents:
t = a["ticker"]
if t in by_co and a["agent_id"] in primary:
by_co[t].append(primary[a["agent_id"]])
out = {}
for ticker, ops in by_co.items():
if not ops:
op = fallback.get(agents[0]["agent_id"]) if agents else None
if op:
ops = [op]
out[ticker] = consensus(ops)
return out, out
def _reinforcement(agents: list[dict], r1: dict[int, dict], r2: dict[int, dict], cfg, plain: bool = False) -> float:
router = route_notes_plain if plain else route_notes
assignments = []
for a in agents:
if a["agent_id"] not in r1:
continue
own = r1[a["agent_id"]]
pool = [{"agent_id": o["agent_id"], "view": o["view"], "confidence": o["confidence"],
"note": o.get("note", ""), "role": o.get("role", "primary")}
for oid, o in r1.items() if oid != a["agent_id"]]
peers = router({**own, "role": a["role"], "agent_id": a["agent_id"]},
pool, k=cfg.peers, seed=cfg.seed)
assignments.append([{"view": own["view"], "agent_id": a["agent_id"]}] + peers)
return round(_reinf_rate(assignments), 4)
def _reinf_rate(assignments: list[list[dict]]) -> float:
total = reinforcing = 0
for asg in assignments:
agent = asg[0]
own = agent["view"]
self_id = agent.get("agent_id")
for note in asg[1:]:
if self_id is not None and note.get("agent_id") == self_id:
continue
total += 1
if note.get("view") == own:
reinforcing += 1
return reinforcing / total if total else 0.0
def _assemble(variant, cfg, eligible, agents, r1, r2, calls, rounds_wall,
before, after, flips, company_consensus, consensus_by_co,
reinforcement, tier_meta=None) -> dict:
pt = sum(c.prompt_tokens for c in calls)
ct = sum(c.completion_tokens for c in calls)
covered = len(company_consensus)
failures = sum(1 for c in calls if not c.ok)
before_valid = [o for o in before if is_valid_opinion(o)]
after_valid = [o for o in after if is_valid_opinion(o)]
return {
"meta": {
"schema_version": SCHEMA_VERSION,
"variant": variant,
"agent_count": len(agents),
"covered_companies": covered,
"eligible_companies": len(eligible),
"not_covered": len(eligible) - covered if variant != "A" else 0,
"seed": cfg.seed,
"n_deepdive": cfg.n_deepdive if variant == "B" else 0,
"tier_meta": tier_meta or {},
},
"opinion_schema": {"score": "0-10", "confidence": "0-1", "views": list(VIEWS)},
"metrics": {
"requests": len(calls),
"failures": failures,
"prompt_tokens": pt, "completion_tokens": ct, "total_tokens": pt + ct,
"cost_per_covered_company": round((pt + ct) / covered, 2) if covered else 0,
"rounds_wall_seconds": [round(w, 3) for w in rounds_wall],
"before_gossip_distribution": distribution(before_valid),
"after_gossip_distribution": distribution(after_valid),
"deepdive_per_agent_flip_rate": round(flips / max(len(before_valid), 1), 4),
"deepdive_flips": flips,
"reinforcement_rate": reinforcement,
"balance_gap_before": _balance_gap(before_valid),
"balance_gap_after": _balance_gap(after_valid),
},
"company_consensus": company_consensus,
"consensus_by_tier": consensus_by_co,
"rounds": {"r1": {str(k): v for k, v in r1.items()},
"r2": {str(k): v for k, v in r2.items()}},
"agents": agents,
}
def _balance_gap(ops: list[dict]) -> float:
n = len(ops) or 1
c = Counter(o["view"] for o in ops if is_valid_opinion(o))
return round(abs(c["bullish"] - c["bearish"]) * 100 / n, 2)
def load_config_from_env(args) -> VariantConfig:
return VariantConfig(
variant=args.variant, agents=args.agents, n_deepdive=args.n_deepdive,
rounds=args.rounds, peers=args.peers, concurrency=args.concurrency,
seed=args.seed, max_retries=args.max_retries, base_delay=args.base_delay,
prior_flags_path=args.prior_flags, output=args.output, dry_run=args.dry_run,
eligible_path=args.eligible, eligible_size=args.eligible_size,
reasoning_effort=args.reasoning_effort,
llm={"base_url": env("LLM_BASE_URL"), "api_key": env("LLM_API_KEY"),
"model": env("LLM_MODEL_NAME"),
"reasoning_effort": args.reasoning_effort or env("LLM_REASONING_EFFORT"),
"prompt_set": args.prompt_set},
)
def main(argv=None) -> int:
p = argparse.ArgumentParser(description="Tiered stock-opinion swarm (rubric v15)")
p.add_argument("--variant", required=True, choices=["A", "B", "C", "control"])
p.add_argument("--prompt-set", choices=["A", "BC"], default="BC",
help="bias control only: which prompt set to test")
p.add_argument("--agents", type=int, default=6000)
p.add_argument("--n-deepdive", type=int, default=200)
p.add_argument("--rounds", type=int, default=2)
p.add_argument("--peers", type=int, default=8)
p.add_argument("--concurrency", type=int, default=256)
p.add_argument("--seed", type=int, default=20260716)
p.add_argument("--max-retries", type=int, default=3)
p.add_argument("--base-delay", type=float, default=0.4)
p.add_argument("--prior-flags", default=None, help="promotion file from a prior B run")
p.add_argument("--eligible", default=None, help="clean eligible dataset JSON")
p.add_argument("--eligible-size", type=int, default=5143)
p.add_argument("--reasoning-effort", default="none")
p.add_argument("--output", default="variant.json")
p.add_argument("--dry-run", action="store_true")
args = p.parse_args(argv)
cfg = load_config_from_env(args)
eff_variant = ("A" if args.prompt_set == "A" else "B") if args.variant == "control" else args.variant
complete = build_complete_fn(cfg.dry_run, cfg.seed, cfg.llm if not cfg.dry_run else None, eff_variant)
out = asyncio.run(run_variant(cfg, complete))
Path(cfg.output).parent.mkdir(parents=True, exist_ok=True)
Path(cfg.output).write_text(json.dumps(out, indent=2))
m = out.get("metrics", {})
if cfg.variant == "control":
sym = m.get("symmetry", {})
print(f"variant=control prompt_set={out['meta'].get('prompt_set')} "
f"sample={out['meta'].get('sample_size')} views={m.get('view_count')} "
f"symmetry_pass={sym.get('pass')} diff={sym.get('diff')} -> {cfg.output}")
else:
print(f"variant={cfg.variant} agents={out['meta']['agent_count']} "
f"covered={out['meta']['covered_companies']} flip={m['deepdive_per_agent_flip_rate']} "
f"reinf={m['reinforcement_rate']} gap_before={m['balance_gap_before']} "
f"tok={m['total_tokens']} -> {cfg.output}")
return 0
if __name__ == "__main__":
sys.exit(main())

258
backend/scripts/winner.py Normal file
View File

@ -0,0 +1,258 @@
#!/usr/bin/env python3
"""Gate 19 winner function for acceptance rubric v15.
Reads variant artifacts (A reference, B tiered, C pure screen, plus optional
bias-control artifacts) and applies the committed, deterministic winner function:
Step 1 calibration gate: B and C must pass gate-13 control + gate-20 calibration.
Step 2 tiered-quality gate: B is eliminated if it fails gate 18 or gate 21.
Step 3 selection: B wins if it survives; otherwise C wins. A never ships.
A is a reference baseline and is never the canonical artifact. Prints the winner
and a per-variant measure table (the gate-26 executable artifact).
"""
from __future__ import annotations
import argparse
import json
import math
import sys
from pathlib import Path
VARIANTS = ("A", "B", "C")
def load(path: str | None) -> dict | None:
if not path or not Path(path).exists():
return None
return json.loads(Path(path).read_text())
def pearson(xs: list[float], ys: list[float]) -> float:
n = len(xs)
if n < 3:
return 0.0
mx = sum(xs) / n
my = sum(ys) / n
num = sum((x - mx) * (y - my) for x, y in zip(xs, ys))
dx = math.sqrt(sum((x - mx) ** 2 for x in xs))
dy = math.sqrt(sum((y - my) ** 2 for y in ys))
if dx == 0 or dy == 0:
return 0.0
return num / (dx * dy)
def fisher_rtoz_test(r_b: float, r_a: float, n: int) -> dict:
"""Fisher r-to-z on the difference; significant if 95% CI of (r_b - r_a) excludes 0."""
if n < 4 or abs(r_b) >= 1 or abs(r_a) >= 1:
return {"significant": False, "ci95": [0.0, 0.0], "reason": "insufficient n or degenerate r"}
zb = 0.5 * math.log((1 + r_b) / (1 - r_b))
za = 0.5 * math.log((1 + r_a) / (1 - r_a))
se = math.sqrt(2 / (n - 3)) # approx for two independent correlations
diff = zb - za
lo, hi = diff - 1.96 * se, diff + 1.96 * se
return {"z_b": round(zb, 4), "z_a": round(za, 4), "diff_z": round(diff, 4),
"ci95": [round(lo, 4), round(hi, 4)], "n": n,
"significant": (lo > 0) or (hi < 0)}
def calibration(variant_art: dict | None) -> dict:
"""Gate 20 + gate 13 control pass flag. Returns pass(bool) and reasons."""
if not variant_art:
return {"pass": False, "reason": "artifact missing"}
# gate 20: agents with confidence > 0.8 must have |score-5| above variant median
ops = []
for r in ("r1", "r2"):
ops.extend(variant_art.get("rounds", {}).get(r, {}).values())
if not ops:
ops = variant_art.get("rounds", {}).get("r1", {}).values()
absd = [abs(float(o["score"]) - 5.0) for o in ops if isinstance(o, dict) and "score" in o]
if not absd:
return {"pass": False, "reason": "no opinions"}
median_absd = sorted(absd)[len(absd) // 2]
hi_conf = [abs(float(o["score"]) - 5.0) for o in ops
if isinstance(o, dict) and float(o.get("confidence", 0)) > 0.8]
violators = sum(1 for d in hi_conf if d < median_absd)
cal_pass = (violators == 0) or (len(hi_conf) == 0)
return {"pass": cal_pass, "median_absd": round(median_absd, 4),
"hi_conf_n": len(hi_conf), "violators": violators,
"reason": "ok" if cal_pass else f"{violators} high-conf agents not more decisive than median"}
def gate13(control_art: dict | None) -> dict:
if not control_art:
return {"pass": False, "reason": "control artifact missing (needs live/bias-control run)"}
sym = control_art["metrics"]["symmetry"]
return {"pass": bool(sym["pass"]), "diff": sym["diff"], "ci95": sym["ci95"],
"n_active": sym["n_active"]}
def deepdive_flip_significance(b: dict, a: dict, dd_tickers: set) -> dict:
"""Two-proportion z-test on per-agent flip rate among deep-dive-company agents."""
def flips_on(art, tickers):
r1, r2 = art["rounds"]["r1"], art.get("rounds", {}).get("r2", {})
agents = art["agents"]
n = f = 0
for ag in agents:
if ag["ticker"] not in tickers:
continue
aid = str(ag["agent_id"])
if aid in r1 and aid in r2:
n += 1
if r1[aid].get("view") != r2[aid].get("view"):
f += 1
return n, f
bn, bf = flips_on(b, dd_tickers)
an, af = flips_on(a, dd_tickers)
if bn == 0 or an == 0:
return {"b_rate": 0, "a_rate": 0, "significant": False, "reason": "no deep-dive agents on A/B"}
br = bf / bn
ar = af / an
# two-proportion z-test
pooled = (bf + af) / (bn + an) if (bn + an) else 0
se = math.sqrt(pooled * (1 - pooled) * (1 / bn + 1 / an)) if pooled not in (0, 1) else 0
z = (br - ar) / se if se > 0 else 0
# one-sided: B higher than A -> significant if z > 1.645 (95% one-sided)
significant = (br > ar) and (z > 1.645)
return {"b_flips": bf, "b_n": bn, "b_rate": round(br, 4),
"a_flips": af, "a_n": an, "a_rate": round(ar, 4),
"z": round(z, 4), "significant": significant}
def gate21(b: dict, a: dict, dd_tickers: set) -> dict:
"""Discrimination r = pearson(consensus confidence, consensus decisive_score_distance)."""
def r_for(art):
cc = art["company_consensus"]
xs, ys = [], []
for t in dd_tickers:
c = cc.get(t)
if not c or c.get("n", 0) == 0:
continue
xs.append(float(c["confidence"]))
ys.append(float(c.get("decisive_score_distance", abs(c["score"] - 5))))
return pearson(xs, ys), len(xs)
r_b, nb = r_for(b)
r_a, na = r_for(a)
sig = fisher_rtoz_test(r_b, r_a, min(nb, na))
return {"r_b": round(r_b, 4), "r_a": round(r_a, 4), "n": min(nb, na),
"significant": sig["significant"], "test": sig,
"pass": (r_b > r_a) and sig["significant"]}
def measure(art: dict) -> dict:
if not art:
return {}
m = art["metrics"]
return {
"covered": art["meta"]["covered_companies"],
"tokens": m["total_tokens"],
"cost_per_covered": m["cost_per_covered_company"],
"reinforcement": m["reinforcement_rate"],
"bal_gap_before": m["balance_gap_before"],
"flip_rate": m["deepdive_per_agent_flip_rate"],
}
def main(argv=None) -> int:
p = argparse.ArgumentParser(description="Gate 19 winner function (rubric v15)")
p.add_argument("--A", required=True)
p.add_argument("--B", required=True)
p.add_argument("--C", required=True)
p.add_argument("--control-A", default=None)
p.add_argument("--control-BC", default=None)
p.add_argument("--output", default=None)
args = p.parse_args(argv)
A, B, C = load(args.A), load(args.B), load(args.C)
cA, cBC = load(args.control_A), load(args.control_BC)
if not (A and B and C):
print("ERROR: missing A/B/C artifact", file=sys.stderr)
return 2
dd = set(B["meta"]["tier_meta"]["deepdive_tickers"])
g13_A = gate13(cA)
g13_BC = gate13(cBC)
cal = {v: calibration({"A": A, "B": B, "C": C}[v]) for v in VARIANTS}
g16 = deepdive_flip_significance(B, A, dd)
g21 = gate21(B, A, dd)
# gate 18: B vs A
reinf_pass = B["metrics"]["reinforcement_rate"] < A["metrics"]["reinforcement_rate"]
balance_pass = B["metrics"]["balance_gap_before"] < A["metrics"]["balance_gap_before"]
g18 = {"reinforcement": reinf_pass, "balance": balance_pass,
"flip": g16["significant"], "pass": reinf_pass and balance_pass and g16["significant"]}
# ---- winner function ----
# step 1: calibration gate
step1 = {
"B": cal["B"]["pass"] and g13_BC["pass"],
"C": cal["C"]["pass"] and g13_BC["pass"],
}
# step 2: tiered-quality gate (B only)
step2_B_pass = g18["pass"] and g21["pass"]
survivors = []
if step1["B"] and step2_B_pass:
survivors.append("B")
if step1["C"]:
survivors.append("C")
# step 3: prefer B; else C
if "B" in survivors:
winner = "B"
elif "C" in survivors:
winner = "C"
else:
winner = None
table = {v: measure(art) for v, art in (("A", A), ("B", B), ("C", C))}
table["A"]["note"] = "reference baseline, not shippable"
if winner is None:
table["verdict"] = "NO SHIP — neither design survived the winner function"
report = {
"winner": winner,
"A_is": "reference baseline, not shippable",
"gate13_A": g13_A, "gate13_BC": g13_BC,
"gate20_calibration": cal,
"gate16_deepdive_flip_vs_A": g16,
"gate18_B_beats_A": g18,
"gate21_discrimination": g21,
"step1_calibration": step1,
"step2_tiered_quality_B_pass": step2_B_pass,
"survivors": survivors,
"measure_table": table,
}
if args.output:
Path(args.output).write_text(json.dumps(report, indent=2))
print("=== GATE 19 WINNER FUNCTION (rubric v15) ===")
print(f"WINNER: {winner if winner else 'NONE'}")
print(f"A = reference baseline, never shippable")
print(f"\n-- Gate 13 (prompt-bias symmetry, control run) --")
print(f" A control : pass={g13_A['pass']} diff={g13_A.get('diff')} ci95={g13_A.get('ci95')}")
print(f" BC control: pass={g13_BC['pass']} diff={g13_BC.get('diff')} ci95={g13_BC.get('ci95')}")
print(f"\n-- Gate 20 (conviction calibration) --")
for v in VARIANTS:
print(f" {v}: pass={cal[v]['pass']} ({cal[v]['reason']})")
print(f"\n-- Gate 16 (deep-dive flip rate vs A, significance) --")
print(f" B flip={g16['b_rate']} (n={g16['b_n']}) A flip={g16['a_rate']} (n={g16['a_n']}) z={g16['z']} significant={g16['significant']}")
print(f"\n-- Gate 18 (B beats A: all three) --")
print(f" reinforcement<{A['metrics']['reinforcement_rate']}? {reinf_pass} balance_gap<{A['metrics']['balance_gap_before']}? {balance_pass} flip significant? {g16['significant']} -> G18 pass={g18['pass']}")
print(f"\n-- Gate 21 (discrimination r, cross-tier, no averaging confound) --")
print(f" r_B={g21['r_b']} r_A={g21['r_a']} n={g21['n']} B>A and significant? {g21['pass']}")
print(f"\n-- Winner-function steps --")
print(f" step1 survivors (calibration): B={step1['B']} C={step1['C']}")
print(f" step2 B tiered-quality pass: {step2_B_pass}")
print(f" survivors: {survivors}")
print(f"\n-- Measure table --")
print(f" {'variant':7} {'covered':>8} {'tokens':>10} {'cost/cov':>9} {'reinf':>7} {'gap_pre':>8} {'flip':>7}")
for v in ("A", "B", "C"):
t = table.get(v, {})
if t:
print(f" {v:7} {t.get('covered',0):>8} {t.get('tokens',0):>10} {t.get('cost_per_covered',0):>9} {t.get('reinforcement',0):>7} {t.get('bal_gap_before',0):>8} {t.get('flip_rate',0):>7}")
print(f"\nCANONICAL ARTIFACT: {winner if winner else 'NONE (no ship)'}")
return 0 if winner else 1
if __name__ == "__main__":
sys.exit(main())

View File

@ -0,0 +1,184 @@
WITH
meta AS
(
SELECT ticker, name, sector, industry, scalemarketcap, exchange, isdelisted
FROM fundamentals.v_companies
WHERE ticker = {ticker:String}
),
annual_periods AS
(
SELECT * EXCEPT rn
FROM
(
SELECT
calendardate, datekey, revenue, netinc, netmargin, epsdil,
grossmargin, roe, roa, marketcap, pe, ps, pb, ev, fcf,
debt, cashneq, equity, ebitda,
row_number() OVER
(
PARTITION BY calendardate
ORDER BY datekey DESC
) AS rn
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRY'
)
WHERE rn = 1
),
latest AS
(
SELECT *
FROM annual_periods
ORDER BY calendardate DESC
LIMIT 1
),
prior AS
(
SELECT revenue, netinc
FROM annual_periods
ORDER BY calendardate DESC
LIMIT 1 OFFSET 1
),
ttm AS
(
SELECT revenue, netinc, epsdil, fcf
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRT'
ORDER BY calendardate DESC, datekey DESC
LIMIT 1
),
growth AS
(
SELECT
round((a.revenue / nullIf(p.revenue, 0) - 1) * 100, 2) AS revenue_yoy_pct,
round((a.netinc / nullIf(p.netinc, 0) - 1) * 100, 2) AS earnings_yoy_pct
FROM latest AS a, prior AS p
),
sector_target AS
(
SELECT pe, ps, roe, sector
FROM fundamentals.v_latest_fundamentals
WHERE ticker = {ticker:String} AND dimension = 'MRY'
LIMIT 1
),
sector_pct AS
(
SELECT
if((SELECT pe FROM sector_target) > 0,
round(countIf(f.pe > 0 AND f.pe <= (SELECT pe FROM sector_target)) * 100.0 /
nullIf(countIf(f.pe > 0), 0), 1),
NULL) AS pe_percentile_in_sector,
if((SELECT ps FROM sector_target) > 0,
round(countIf(f.ps > 0 AND f.ps <= (SELECT ps FROM sector_target)) * 100.0 /
nullIf(countIf(f.ps > 0), 0), 1),
NULL) AS ps_percentile_in_sector,
round(countIf(f.roe IS NOT NULL AND f.roe <= (SELECT roe FROM sector_target)) * 100.0 /
nullIf(countIf(f.roe IS NOT NULL), 0), 1) AS roe_percentile_in_sector
FROM fundamentals.v_latest_fundamentals AS f
WHERE f.dimension = 'MRY'
AND f.sector = (SELECT sector FROM sector_target)
),
quarterly_periods AS
(
SELECT * EXCEPT rn
FROM
(
SELECT
calendardate, datekey, fiscalperiod, revenue, netinc, eps, grossmargin,
row_number() OVER
(
PARTITION BY calendardate
ORDER BY datekey DESC
) AS rn
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRQ'
)
WHERE rn = 1
ORDER BY calendardate DESC
LIMIT 8
),
holders AS
(
SELECT investorname, units AS shares, value AS usd_value
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate = {holding_quarter:Date}
ORDER BY value DESC
LIMIT 10
),
concentration AS
(
SELECT
round(sum(value), 0) AS total_value,
countDistinct(investorname) AS holder_count,
round(
(
SELECT sum(v)
FROM
(
SELECT value AS v
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate = {holding_quarter:Date}
ORDER BY value DESC
LIMIT 5
)
) * 100.0 / nullIf(sum(value), 0),
1
) AS top5_concentration_pct
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate = {holding_quarter:Date}
),
ownership_trend AS
(
SELECT calendardate, round(sum(value), 0) AS total_value,
countDistinct(investorname) AS holder_count
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate <= {holding_quarter:Date}
GROUP BY calendardate
ORDER BY calendardate DESC
LIMIT 4
)
SELECT
{ticker:String} AS ticker,
(SELECT name FROM meta) AS name,
(SELECT sector FROM meta) AS sector,
(SELECT industry FROM meta) AS industry,
(SELECT scalemarketcap FROM meta) AS scale_marketcap,
(SELECT exchange FROM meta) AS exchange,
(SELECT isdelisted FROM meta) AS is_delisted,
(SELECT calendardate FROM latest) AS latest_annual_date,
(SELECT revenue FROM latest) AS revenue_annual,
(SELECT netinc FROM latest) AS net_income_annual,
(SELECT revenue FROM ttm) AS revenue_ttm,
(SELECT netinc FROM ttm) AS net_income_ttm,
(SELECT epsdil FROM ttm) AS eps_diluted_ttm,
(SELECT fcf FROM ttm) AS free_cash_flow_ttm,
(SELECT netmargin FROM latest) AS net_margin,
(SELECT grossmargin FROM latest) AS gross_margin,
(SELECT roe FROM latest) AS roe,
(SELECT marketcap FROM latest) AS market_cap,
(SELECT pe FROM latest) AS pe,
(SELECT ps FROM latest) AS ps,
(SELECT pb FROM latest) AS pb,
(SELECT ev FROM latest) AS enterprise_value,
(SELECT debt FROM latest) AS debt,
(SELECT cashneq FROM latest) AS cash,
(SELECT equity FROM latest) AS equity,
(SELECT ebitda FROM latest) AS ebitda,
(SELECT revenue_yoy_pct FROM growth) AS revenue_yoy_pct,
(SELECT earnings_yoy_pct FROM growth) AS earnings_yoy_pct,
(SELECT pe_percentile_in_sector FROM sector_pct) AS pe_percentile_in_sector,
(SELECT ps_percentile_in_sector FROM sector_pct) AS ps_percentile_in_sector,
(SELECT roe_percentile_in_sector FROM sector_pct) AS roe_percentile_in_sector,
(SELECT total_value FROM concentration) AS institutional_total_value,
(SELECT holder_count FROM concentration) AS institutional_holder_count,
(SELECT top5_concentration_pct FROM concentration) AS institutional_top5_concentration_pct,
(SELECT groupArray((calendardate, fiscalperiod, revenue, netinc, eps, grossmargin)) FROM quarterly_periods) AS quarterly_trend,
(SELECT groupArray((investorname, shares, usd_value)) FROM holders) AS top_holders,
(SELECT groupArray((calendardate, total_value, holder_count)) FROM ownership_trend) AS ownership_trend

View File

@ -0,0 +1,166 @@
WITH
meta AS
(
SELECT ticker, name, sector, industry, scalemarketcap, exchange, isdelisted
FROM fundamentals.v_companies
WHERE ticker = {ticker:String}
),
annual_periods AS
(
SELECT * EXCEPT rn
FROM
(
SELECT
calendardate, datekey, revenue, netinc, netmargin, epsdil,
grossmargin, roe, roa, marketcap, pe, ps, pb, ev, fcf,
debt, cashneq, equity, ebitda,
row_number() OVER
(
PARTITION BY calendardate
ORDER BY datekey DESC
) AS rn
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRY'
)
WHERE rn = 1
),
latest AS
(
SELECT *
FROM annual_periods
ORDER BY calendardate DESC
LIMIT 1
),
prior AS
(
SELECT revenue, netinc
FROM annual_periods
ORDER BY calendardate DESC
LIMIT 1 OFFSET 1
),
ttm AS
(
SELECT revenue, netinc, epsdil, fcf
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRT'
ORDER BY calendardate DESC, datekey DESC
LIMIT 1
),
growth AS
(
SELECT
round((a.revenue / nullIf(p.revenue, 0) - 1) * 100, 2) AS revenue_yoy_pct,
round((a.netinc / nullIf(p.netinc, 0) - 1) * 100, 2) AS earnings_yoy_pct
FROM latest AS a, prior AS p
),
sector_target AS
(
SELECT pe, ps, roe, sector
FROM fundamentals.v_latest_fundamentals
WHERE ticker = {ticker:String} AND dimension = 'MRY'
LIMIT 1
),
sector_pct AS
(
SELECT
if((SELECT pe FROM sector_target) > 0,
round(countIf(f.pe > 0 AND f.pe <= (SELECT pe FROM sector_target)) * 100.0 /
nullIf(countIf(f.pe > 0), 0), 1),
NULL) AS pe_percentile_in_sector,
if((SELECT ps FROM sector_target) > 0,
round(countIf(f.ps > 0 AND f.ps <= (SELECT ps FROM sector_target)) * 100.0 /
nullIf(countIf(f.ps > 0), 0), 1),
NULL) AS ps_percentile_in_sector,
round(countIf(f.roe IS NOT NULL AND f.roe <= (SELECT roe FROM sector_target)) * 100.0 /
nullIf(countIf(f.roe IS NOT NULL), 0), 1) AS roe_percentile_in_sector
FROM fundamentals.v_latest_fundamentals AS f
WHERE f.dimension = 'MRY'
AND f.sector = (SELECT sector FROM sector_target)
),
latest_holding_quarter AS
(
SELECT max(calendardate) AS q
FROM fundamentals.sf3
WHERE ticker = {ticker:String} AND securitytype = 'SHR'
),
concentration AS
(
SELECT
round(sum(value), 0) AS total_value,
countDistinct(investorname) AS holder_count,
round(
(
SELECT sum(v)
FROM
(
SELECT value AS v
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate = (SELECT q FROM latest_holding_quarter)
ORDER BY value DESC
LIMIT 5
)
) * 100.0 / nullIf(sum(value), 0),
1
) AS top5_concentration_pct
FROM fundamentals.sf3
WHERE ticker = {ticker:String}
AND securitytype = 'SHR'
AND calendardate = (SELECT q FROM latest_holding_quarter)
),
quarterly_periods AS
(
SELECT * EXCEPT rn
FROM
(
SELECT
calendardate, datekey, fiscalperiod, revenue, netinc, eps, grossmargin,
row_number() OVER
(
PARTITION BY calendardate
ORDER BY datekey DESC
) AS rn
FROM fundamentals.sf1
WHERE ticker = {ticker:String} AND dimension = 'MRQ'
)
WHERE rn = 1
ORDER BY calendardate DESC
LIMIT 8
)
SELECT
{ticker:String} AS ticker,
(SELECT name FROM meta) AS name,
(SELECT sector FROM meta) AS sector,
(SELECT industry FROM meta) AS industry,
(SELECT scalemarketcap FROM meta) AS scale_marketcap,
(SELECT exchange FROM meta) AS exchange,
(SELECT isdelisted FROM meta) AS is_delisted,
(SELECT calendardate FROM latest) AS latest_annual_date,
(SELECT revenue FROM latest) AS revenue_annual,
(SELECT netinc FROM latest) AS net_income_annual,
(SELECT revenue FROM ttm) AS revenue_ttm,
(SELECT netinc FROM ttm) AS net_income_ttm,
(SELECT epsdil FROM ttm) AS eps_diluted_ttm,
(SELECT fcf FROM ttm) AS free_cash_flow_ttm,
(SELECT netmargin FROM latest) AS net_margin,
(SELECT grossmargin FROM latest) AS gross_margin,
(SELECT roe FROM latest) AS roe,
(SELECT marketcap FROM latest) AS market_cap,
(SELECT pe FROM latest) AS pe,
(SELECT ps FROM latest) AS ps,
(SELECT pb FROM latest) AS pb,
(SELECT ev FROM latest) AS enterprise_value,
(SELECT debt FROM latest) AS debt,
(SELECT cashneq FROM latest) AS cash,
(SELECT equity FROM latest) AS equity,
(SELECT ebitda FROM latest) AS ebitda,
(SELECT revenue_yoy_pct FROM growth) AS revenue_yoy_pct,
(SELECT earnings_yoy_pct FROM growth) AS earnings_yoy_pct,
(SELECT pe_percentile_in_sector FROM sector_pct) AS pe_percentile_in_sector,
(SELECT ps_percentile_in_sector FROM sector_pct) AS ps_percentile_in_sector,
(SELECT roe_percentile_in_sector FROM sector_pct) AS roe_percentile_in_sector,
(SELECT total_value FROM concentration) AS institutional_total_value,
(SELECT holder_count FROM concentration) AS institutional_holder_count,
(SELECT top5_concentration_pct FROM concentration) AS institutional_top5_concentration_pct,
(SELECT groupArray((calendardate, fiscalperiod, revenue, netinc, eps, grossmargin)) FROM quarterly_periods) AS quarterly_trend

View File

@ -0,0 +1,69 @@
import adversarial_router as ar
def _agent(aid: str, role: str, view: str) -> dict:
return {"agent_id": aid, "role": role, "view": view,
"score": 5.0, "confidence": 0.7, "thesis": "t", "note": "n"}
def _balanced_agents() -> list[dict]:
agents = [_agent(f"bull-{i}", ar.BULL_LEANING[i % 3], "bullish") for i in range(8)]
agents += [_agent(f"bear-{i}", ar.BEAR_LEANING[i % 3], "bearish") for i in range(8)]
return agents
def test_classify_agent():
assert ar.classify_agent("growth") == "bull"
assert ar.classify_agent("risk") == "bear"
assert ar.classify_agent("macro") == "neutral"
def test_balanced_fixture_low_reinforcement():
pool = ar.synthetic_bear_market_pool(8, 8, 0, seed=0) # 50% bull, 50% bear notes
agents = _balanced_agents() # 8 bull-leaning + 8 bear-leaning
assignments = [[a, *ar.route_notes(a, pool, k=8, seed=0)] for a in agents]
rate = ar.reinforcement_rate(assignments)
assert rate < 0.10, f"reinforcement too high: {rate}"
def test_routing_is_deterministic():
pool = ar.synthetic_bear_market_pool(6, 6, 4, seed=1)
agent = _agent("x", "growth", "bullish")
a = ar.route_notes(agent, pool, k=8, seed=42)
b = ar.route_notes(agent, pool, k=8, seed=42)
assert a == b, "same seed + agent must yield identical routing"
def test_same_view_only_when_no_opposite_or_neutral():
agent = _agent("self", "growth", "bullish")
all_bull = [{"agent_id": f"b{i}", "view": "bullish", "score": 5,
"confidence": 0.5, "thesis": "t", "note": "n"} for i in range(5)]
out = ar.route_notes(agent, all_bull, k=8, seed=0)
assert len(out) > 0, "must fall back to same-view when nothing else exists"
assert all(n["view"] == "bullish" for n in out)
assert ar.reinforcement_rate([[agent, *out]]) == 1.0 # same-view counts
# once an opposite note appears, same-view must disappear
pool = all_bull + [{"agent_id": "bear1", "view": "bearish", "score": 5,
"confidence": 0.5, "thesis": "t", "note": "n"}]
out2 = ar.route_notes(agent, pool, k=8, seed=0)
assert len(out2) > 0
assert all(n["view"] == "bearish" for n in out2), "opposite exists -> no same-view"
def test_self_note_never_returned():
pool = ar.synthetic_bear_market_pool(5, 5, 2, seed=3)
agent = _agent(pool[0]["agent_id"], "quality", "bearish")
out = ar.route_notes(agent, pool, k=8, seed=0)
assert out, "expected some notes"
assert all(n.get("agent_id") != agent["agent_id"] for n in out)
def test_dedup_holds():
base = {"view": "bearish", "score": 5, "confidence": 0.5, "thesis": "t", "note": "n"}
pool = [{**base, "agent_id": "dup"}, {**base, "agent_id": "dup"},
{**base, "agent_id": "dup"}, {**base, "agent_id": "other"}]
agent = _agent("self", "growth", "bullish")
out = ar.route_notes(agent, pool, k=8, seed=0)
ids = [n.get("agent_id") for n in out]
assert len(ids) == len(set(ids)), "duplicate agent_ids returned"

View File

@ -0,0 +1,89 @@
import math
import bias_control as bc
def _dossiers():
return [
{"ticker": "AAA", "sector": "tech", "revenue_annual": 100, "roe": 0.1, "pe": 20, "ps": 4},
{"ticker": "BBB", "sector": "tech", "revenue_annual": 300, "roe": 0.3, "pe": 40, "ps": 8},
{"ticker": "CCC", "sector": "tech", "revenue_annual": 200, "roe": 0.2, "pe": 30, "ps": 6},
{"ticker": "DDD", "sector": "energy", "revenue_annual": 50, "roe": None, "pe": 10, "ps": 2},
{"ticker": "EEE", "sector": "energy", "revenue_annual": 150, "roe": 0.15, "pe": 15, "ps": 3},
]
def test_sector_median_ignores_none_and_returns_medians():
medians = bc.sector_median_fundamentals(_dossiers())
assert set(medians) == {"tech", "energy"}
# tech sorted revenue 100,200,300 -> median 200; roe 0.1,0.2,0.3 -> 0.2
assert medians["tech"]["revenue_annual"] == 200
assert medians["tech"]["roe"] == 0.2
# energy roe had one None -> only 0.15 survives -> median 0.15
assert medians["energy"]["roe"] == 0.15
def test_synthetic_neutral_dossier_pins_medians_and_marks_flag():
medians = bc.sector_median_fundamentals(_dossiers())
d = bc.synthetic_neutral_dossier("ZZZ", "tech", medians)
assert d["ticker"] == "ZZZ"
assert d["sector"] == "tech"
assert d["synthetic_neutral"] is True
assert d["revenue_annual"] == medians["tech"]["revenue_annual"] == 200
assert d["roe"] == 0.2
# field absent from medians stays None
assert d["net_margin"] is None
def test_deterministic_sample_reproducible():
items = list(range(1000))
a = bc.deterministic_sample(items, 500, 7)
b = bc.deterministic_sample(items, 500, 7)
assert a == b
assert len(a) == 500
def test_deterministic_sample_clamps_to_min_500():
items = list(range(600))
assert len(bc.deterministic_sample(items, 10, 1)) == 500
def test_deterministic_sample_returns_all_when_fewer_than_500():
items = list(range(30))
out = bc.deterministic_sample(items, 10, 1)
assert len(out) == 30
assert sorted(out) == items
def test_mcnemar_symmetric_passes_and_ci_contains_zero():
views = ["bullish"] * 100 + ["bearish"] * 100 + ["neutral"] * 50
r = bc.mcnemar_symmetry(views)
assert r["n_bull"] == 100 and r["n_bear"] == 100 and r["n_neutral"] == 50
assert r["n_active"] == 200
assert r["diff"] == 0.0
lo, hi = r["ci95"]
assert lo < 0 < hi
assert r["pass"] is True
def test_mcnemar_skewed_fails():
views = ["bullish"] * 300 + ["bearish"] * 10
r = bc.mcnemar_symmetry(views)
assert r["n_active"] == 310
assert r["diff"] < 0
lo, hi = r["ci95"]
assert hi < 0
assert r["pass"] is False
def test_run_bias_control_end_to_end():
companies = [
{"ticker": f"T{i}", "sector": "tech", "revenue_annual": i * 10, "roe": i / 100}
for i in range(600)
]
# balanced views -> no bearish bias detected
views = (["bullish", "bearish"] * 250) + ["neutral"] * 100
out = bc.run_bias_control(companies, views, n=500, seed=13)
assert out["n_sampled"] == 500
assert out["symmetry"]["pass"] is True
assert all(d["synthetic_neutral"] for d in out["dossiers"][:5])

View File

@ -0,0 +1,55 @@
import re
from pathlib import Path
SQL_DIR = Path(__file__).resolve().parent.parent / "sql"
BASE_27 = [
"ticker", "name", "sector", "industry", "scale_marketcap", "exchange",
"is_delisted", "latest_annual_date", "revenue_annual", "net_income_annual",
"revenue_ttm", "net_income_ttm", "eps_diluted_ttm", "free_cash_flow_ttm",
"net_margin", "gross_margin", "roe", "market_cap", "pe", "ps", "pb",
"enterprise_value", "debt", "cash", "equity", "ebitda", "revenue_yoy_pct",
"earnings_yoy_pct", "pe_percentile_in_sector", "ps_percentile_in_sector",
"roe_percentile_in_sector", "institutional_total_value",
"institutional_holder_count", "institutional_top5_concentration_pct",
]
def _final_select(sql: str) -> str:
m = re.search(r"(?m)^SELECT$", sql)
assert m, "no top-level SELECT found; file does not parse as a CTE query"
return sql[m.start():]
def _has_alias(final_select: str, alias: str) -> bool:
return re.search(rf"\bAS\s+{re.escape(alias)}\b", final_select, re.IGNORECASE) is not None
def test_screen_selects_base_27_and_quarterly_trend():
sql = (SQL_DIR / "dossier_screen.sql").read_text()
fs = _final_select(sql)
for alias in BASE_27:
assert _has_alias(fs, alias), f"screen missing alias: {alias}"
assert _has_alias(fs, "quarterly_trend"), "screen missing quarterly_trend"
assert "{ticker:String}" in sql, "screen must be parameterized by ticker"
assert "{holding_quarter" not in sql, "screen must not reference holding_quarter"
def test_deepdive_is_strict_superset_of_screen():
screen_fs = _final_select((SQL_DIR / "dossier_screen.sql").read_text())
deepdive_fs = _final_select((SQL_DIR / "dossier_deepdive.sql").read_text())
screen_aliases = set(re.findall(r"\bAS\s+([A-Za-z_][A-Za-z0-9_]*)\b", screen_fs, re.IGNORECASE))
assert screen_aliases, "no aliases found in screen final SELECT"
for alias in screen_aliases:
assert _has_alias(deepdive_fs, alias), f"deepdive missing screen alias: {alias}"
for extra in ("top_holders", "ownership_trend"):
assert _has_alias(deepdive_fs, extra), f"deepdive missing extra alias: {extra}"
def test_param_arity():
screen = (SQL_DIR / "dossier_screen.sql").read_text()
deepdive = (SQL_DIR / "dossier_deepdive.sql").read_text()
assert "{ticker:String}" in screen and "{holding_quarter" not in screen, \
"screen must use only {ticker:String}"
assert "{ticker:String}" in deepdive and "{holding_quarter:Date}" in deepdive, \
"deepdive must use both {ticker:String} and {holding_quarter:Date}"

View File

@ -0,0 +1,261 @@
"""Tests for the tiered runner + winner function (rubric v15)."""
import json
import os
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "scripts"))
import run_tiered_swarm as rt # noqa: E402
import winner as wn # noqa: E402
# ---- tier assignment (gates 3, 17, 25) ----
def _elig(n):
return [{"ticker": f"T{i:04d}", "name": f"Co {i}", "sector": "Tech",
"market_cap": 10_000 - i} for i in range(n)]
def test_assign_tiers_budget_and_no_alpha():
elig = _elig(20)
deep, screen = rt.assign_tiers(elig, n_deepdive=3, prior_flags=None)
assert len(deep) == 3
# top by market cap (descending) — not alphabetical
assert [d["ticker"] for d in deep] == ["T0000", "T0001", "T0002"]
assert all(d["ticker"] not in {s["ticker"] for s in screen} for d in deep)
def test_assign_tires_prior_flags_promoted_first():
elig = _elig(20)
deep, screen = rt.assign_tiers(elig, n_deepdive=3, prior_flags=["T0019"])
assert "T0019" in [d["ticker"] for d in deep]
def test_assign_tiers_deterministic():
elig = _elig(50)
d1, s1 = rt.assign_tiers(elig, 5, None)
d2, s2 = rt.assign_tiers(elig, 5, None)
assert [d["ticker"] for d in d1] == [d["ticker"] for d in d2]
def test_budget_arithmetic_covered_not_covered():
n_deepdive = 200
agents = 6000
covered_per_b = n_deepdive + (agents - 6 * n_deepdive) # = 6000 - 5N
assert covered_per_b == 6000 - 5 * n_deepdive == 5000
# ---- consensus tie-break (gate 7) ----
def test_consensus_tiebreak_confidence_then_agent_id():
a = {"view": "bullish", "score": 7, "confidence": 0.8, "agent_id": 5}
b = {"view": "bullish", "score": 7, "confidence": 0.8, "agent_id": 2}
c = {"view": "bearish", "score": 3, "confidence": 0.9, "agent_id": 9}
cons = rt.consensus([a, b, c])
assert cons["view"] == "bullish" # 2 vs 1
assert cons["winner_agent_id"] == 2 # higher conf, then smaller id
def test_consensus_view_tiebreak_by_confidence():
a = {"view": "bullish", "score": 7, "confidence": 0.6, "agent_id": 1}
b = {"view": "bearish", "score": 3, "confidence": 0.95, "agent_id": 2}
cons = rt.consensus([a, b])
# one each -> the view with higher mean confidence wins (bearish)
assert cons["view"] == "bearish"
def test_consensus_empty_is_neutral():
cons = rt.consensus([])
assert cons["view"] == "neutral" and cons["n"] == 0
# ---- promotion (gate 24) ----
def test_promotion_flags_high_conviction_non_neutral():
cc = {"A": {"view": "bullish", "score": 9, "confidence": 0.9},
"B": {"view": "neutral", "score": 5, "confidence": 0.5},
"C": {"view": "bearish", "score": 2, "confidence": 0.7}} # not sharp enough
flags = rt.promotion_flags(cc)
assert "A" in flags and "B" not in flags and "C" not in flags
def test_promotion_rule_does_not_read_own_screen_for_deepdive_set():
# the assignment-time rule uses market cap + prior flags only — coverage by
# the assignment function which never sees opinions
elig = _elig(30)
deep, _ = rt.assign_tiers(elig, 4, None)
# determinism + market-cap-driven (not opinion-driven)
assert deep[0]["market_cap"] == 10000
# ---- balance gap (gate 18) ----
def test_balance_gap():
ops = [{"view": v, "score": s, "confidence": 0.5} for v, s in
[("bullish", 8), ("bullish", 8), ("bullish", 8), ("bearish", 2), ("neutral", 5)]]
assert rt._balance_gap(ops) == 40.0 # |3-1|/5*100
# ---- winner function (gate 19) on fixtures ----
def _op(view, score, conf, aid):
return {"view": view, "score": score, "confidence": conf, "agent_id": aid}
def _write_artifact(tmp, name, deepdive_tickers=None, r1=None, r2=None,
agents=None, company_consensus=None, metrics=None):
deepdive_tickers = deepdive_tickers or []
r1 = r1 or {}
r2 = r2 or {}
agents = agents or []
company_consensus = company_consensus or {}
metrics = metrics or {"total_tokens": 1000, "cost_per_covered_company": 10,
"balance_gap_before": 0.0, "reinforcement_rate": 0.0,
"deepdive_per_agent_flip_rate": 0.0,
"balance_gap_after": 0.0}
art = {
"meta": {"covered_companies": len(company_consensus),
"agent_count": len(agents),
"tier_meta": {"deepdive_tickers": deepdive_tickers}},
"metrics": metrics,
"company_consensus": company_consensus,
"rounds": {"r1": {str(k): v for k, v in r1.items()},
"r2": {str(k): v for k, v in r2.items()}},
"agents": agents,
}
p = tmp / name
p.write_text(json.dumps(art))
return str(p)
def _control(pass_, diff):
p = {
"meta": {"prompt_set": "X", "sample_size": 500},
"metrics": {"requests": 500, "view_count": 500,
"symmetry": {"pass": pass_, "diff": diff, "ci95": [0, 0],
"n_active": 500}},
"views": [],
}
return p
def test_winner_picks_B_when_B_proves_quality(tmp_path, monkeypatch):
# 5 deep-dive companies with varying decisive score distance.
scores = [6, 7, 8, 9, 10]
decisive = [abs(s - 5) for s in scores] # [1,2,3,4,5]
dd = [f"CO{i}" for i in range(5)]
# A: naive single agents — confidence ~constant regardless of decisiveness (low r)
A_cc = {t: {"view": "bullish", "score": s, "confidence": 0.55,
"decisive_score_distance": d, "n": 1}
for t, s, d in zip(dd, scores, decisive)}
a_agents = [{"agent_id": i, "ticker": dd[i]} for i in range(5)]
a_r1 = {i: _op("bullish", scores[i], 0.55, i) for i in range(5)}
a_r2 = a_r1 # no flips for A
A = _write_artifact(tmp_path, "A.json", dd, a_r1, a_r2, a_agents, A_cc,
metrics={"total_tokens": 2000, "cost_per_covered_company": 400,
"balance_gap_before": 50.0, "reinforcement_rate": 0.9,
"deepdive_per_agent_flip_rate": 0.0,
"balance_gap_after": 50.0})
# B: debated agents — confidence tracks decisiveness (high r, not exactly 1.0);
# agents flip r1->r2. r2 confidence per company tracks decisive so gate 20 holds.
b_conf = [0.42, 0.54, 0.66, 0.78, 0.85] # increases with decisive
B_cc = {t: {"view": "bullish", "score": s, "confidence": c,
"decisive_score_distance": d, "n": 1}
for t, s, c, d in zip(dd, scores, b_conf, decisive)}
b_agents = [{"agent_id": 100 + i, "ticker": dd[i // 6]} for i in range(30)] # 6 per company
b_r1 = {100 + i: _op("bearish", 5 - decisive[i // 6], 0.5, 100 + i) for i in range(30)}
b_r2 = {100 + i: _op("bullish", scores[i // 6], b_conf[i // 6], 100 + i) for i in range(30)}
B = _write_artifact(tmp_path, "B.json", dd, b_r1, b_r2, b_agents, B_cc,
metrics={"total_tokens": 1500, "cost_per_covered_company": 300,
"balance_gap_before": 2.0, "reinforcement_rate": 0.05,
"deepdive_per_agent_flip_rate": 1.0,
"balance_gap_after": 2.0})
Ccc = {t: {"view": "bullish", "score": s, "confidence": 0.55,
"decisive_score_distance": d, "n": 1} for t, s, d in zip(dd, scores, decisive)}
c_agents = [{"agent_id": i, "ticker": dd[i]} for i in range(5)]
c_r1 = {i: _op("bullish", scores[i], 0.55, i) for i in range(5)}
C = _write_artifact(tmp_path, "C.json", dd, c_r1, {}, c_agents, Ccc,
metrics={"total_tokens": 800, "cost_per_covered_company": 160,
"balance_gap_before": 0.0, "reinforcement_rate": 0.0,
"deepdive_per_agent_flip_rate": 0.0,
"balance_gap_after": 0.0})
cA = tmp_path / "cA.json"; cA.write_text(json.dumps(_control(False, 1.0)))
cBC = tmp_path / "cBC.json"; cBC.write_text(json.dumps(_control(True, 0.0)))
out = tmp_path / "out.json"
rc = wn.main(["--A", A, "--B", B, "--C", C, "--control-A", str(cA),
"--control-BC", str(cBC), "--output", str(out)])
report = json.loads(out.read_text())
assert rc == 0
assert report["winner"] == "B"
assert report["gate18_B_beats_A"]["pass"] is True
assert report["gate21_discrimination"]["r_b"] > report["gate21_discrimination"]["r_a"]
assert report["gate21_discrimination"]["pass"] is True
def test_winner_falls_back_to_C_when_B_fails_quality(tmp_path):
dd = ["CO1"]
A_cc = {"CO1": {"view": "bearish", "score": 2, "confidence": 0.55,
"decisive_score_distance": 3.0, "n": 1}}
a = [{"agent_id": 0, "ticker": "CO1"}]
r1 = {0: _op("bearish", 2, 0.55, 0)}
A = _write_artifact(tmp_path, "A.json", dd, r1, r1, a, A_cc,
metrics={"total_tokens": 1000, "cost_per_covered_company": 1000,
"balance_gap_before": 10.0, "reinforcement_rate": 0.1,
"deepdive_per_agent_flip_rate": 0.5,
"balance_gap_after": 10.0})
# B fails gate 21: r_B not > r_A (same correlation)
B_cc = {"CO1": {"view": "bullish", "score": 8, "confidence": 0.9,
"decisive_score_distance": 3.0, "n": 1}}
B = _write_artifact(tmp_path, "B.json", dd, r1, r1, a, B_cc,
metrics={"total_tokens": 1000, "cost_per_covered_company": 500,
"balance_gap_before": 1.0, "reinforcement_rate": 0.0,
"deepdive_per_agent_flip_rate": 0.9,
"balance_gap_after": 1.0})
C = _write_artifact(tmp_path, "C.json", dd, r1, {}, a, B_cc,
metrics={"total_tokens": 500, "cost_per_covered_company": 250,
"balance_gap_before": 0.0, "reinforcement_rate": 0.0,
"deepdive_per_agent_flip_rate": 0.0,
"balance_gap_after": 0.0})
cA = tmp_path / "cA.json"; cA.write_text(json.dumps(_control(False, 1.0)))
cBC = tmp_path / "cBC.json"; cBC.write_text(json.dumps(_control(True, 0.0)))
out = tmp_path / "out.json"
rc = wn.main(["--A", A, "--B", B, "--C", C, "--control-A", str(cA),
"--control-BC", str(cBC), "--output", str(out)])
report = json.loads(out.read_text())
assert report["winner"] == "C"
assert "B" not in report["survivors"]
def test_winner_no_ship_when_both_fail_calibration(tmp_path):
dd = ["CO1"]
a = [{"agent_id": 0, "ticker": "CO1"}]
r1 = {0: _op("bearish", 2, 0.99, 0)} # high conf but score close-ish -> calibration fail
cc = {"CO1": {"view": "bearish", "score": 2, "confidence": 0.99,
"decisive_score_distance": 3.0, "n": 1}}
A = _write_artifact(tmp_path, "A.json", dd, r1, r1, a, cc,
metrics={"total_tokens": 1, "cost_per_covered_company": 1,
"balance_gap_before": 50, "reinforcement_rate": 0.9,
"deepdive_per_agent_flip_rate": 0,
"balance_gap_after": 50})
B = _write_artifact(tmp_path, "B.json", dd, r1, r1, a, cc,
metrics={"total_tokens": 1, "cost_per_covered_company": 1,
"balance_gap_before": 1, "reinforcement_rate": 0,
"deepdive_per_agent_flip_rate": 0.9,
"balance_gap_after": 1})
C = _write_artifact(tmp_path, "C.json", dd, r1, {}, a, cc,
metrics={"total_tokens": 1, "cost_per_covered_company": 1,
"balance_gap_before": 0, "reinforcement_rate": 0,
"deepdive_per_agent_flip_rate": 0,
"balance_gap_after": 0})
# both controls fail -> B and C eliminated at step 1
cA = tmp_path / "cA.json"; cA.write_text(json.dumps(_control(False, 1.0)))
cBC = tmp_path / "cBC.json"; cBC.write_text(json.dumps(_control(False, 1.0)))
out = tmp_path / "out.json"
rc = wn.main(["--A", A, "--B", B, "--C", C, "--control-A", str(cA),
"--control-BC", str(cBC), "--output", str(out)])
report = json.loads(out.read_text())
assert report["winner"] is None
assert rc == 1