{ "corpus": "contradictions", "results": [ { "id": "flu-1918-deaths", "question": "how many people died in the 1918 influenza pandemic?", "tags": [ "disputed", "epidemiology", "numeric" ], "session_id": "s_06FZA17P8XNM1NHD", "status": "done", "claims": 36, "scorecard": { "session_id": "s_06FZA17P8XNM1NHD", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0110 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "47 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "36 of 36 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 22.22222222222222, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 36 claims (https://archive.cdc.gov/www_cdc_gov/flu/pandemic-resources/1918-commemoration/1918-pandemic-history.htm)" }, { "name": "cost per claim", "status": "measured", "value": 0.000307, "unit": "usd", "detail": "$0.0110 over 36 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 47, "unit": "calls", "detail": "llm 46 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "36 of 36 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 33.333333333333336, "unit": "%", "detail": "36 claim(s) render as 24 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 16.666666666666668, "unit": "%", "detail": "4 contradiction(s) involving 6 of 36 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 4 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "population-peak", "question": "when will the world population peak and at what level?", "tags": [ "disputed", "demography", "numeric" ], "session_id": "s_06FZA1N1KDMQ4MCT", "status": "done", "claims": 63, "scorecard": { "session_id": "s_06FZA1N1KDMQ4MCT", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0214 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "88 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "63 of 63 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 12.698412698412698, "unit": "%", "detail": "12 distinct source(s); the largest contributed 8 of 63 claims (https://populationmatters.org/news/2024/04/the-world-of-population-projections)" }, { "name": "cost per claim", "status": "measured", "value": 0.000339, "unit": "usd", "detail": "$0.0214 over 63 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 88, "unit": "calls", "detail": "llm 86 · search 2" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "63 of 63 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 31.746031746031747, "unit": "%", "detail": "63 claim(s) render as 43 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 6.349206349206349, "unit": "%", "detail": "2 contradiction(s) involving 4 of 63 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 2 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "bitcoin-energy", "question": "how much electricity does the bitcoin network consume per year?", "tags": [ "disputed", "energy", "numeric" ], "session_id": "s_06FZA2H89XHN6YHJ", "status": "done", "claims": 26, "scorecard": { "session_id": "s_06FZA2H89XHN6YHJ", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0137 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "36 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "26 of 26 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 30.76923076923077, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 26 claims (https://cryptoslate.com/stop-shaming-bitcoin-when-daily-streaming-ai-and-social-media-scrolling-shown-to-consume-double-the-energy)" }, { "name": "cost per claim", "status": "measured", "value": 0.000526, "unit": "usd", "detail": "$0.0137 over 26 claims = $0.0005 each" }, { "name": "tool calls", "status": "measured", "value": 36, "unit": "calls", "detail": "llm 35 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "26 of 26 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 3.8461538461538463, "unit": "%", "detail": "26 claim(s) render as 25 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 23.076923076923077, "unit": "%", "detail": "5 contradiction(s) involving 6 of 26 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 5 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "psych-replication", "question": "what fraction of psychology studies replicate?", "tags": [ "disputed", "metascience", "numeric" ], "session_id": "s_06FZA2X9VRJFF8QZ", "status": "done", "claims": 44, "scorecard": { "session_id": "s_06FZA2X9VRJFF8QZ", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0141 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "63 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "44 of 44 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 18.181818181818183, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 44 claims (https://en.wikipedia.org/wiki/Replication_crisis)" }, { "name": "cost per claim", "status": "measured", "value": 0.00032, "unit": "usd", "detail": "$0.0141 over 44 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 63, "unit": "calls", "detail": "llm 62 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "44 of 44 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 25, "unit": "%", "detail": "44 claim(s) render as 33 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 22.727272727272727, "unit": "%", "detail": "7 contradiction(s) involving 10 of 44 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 7 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "nuclear-vs-solar-lcoe", "question": "is nuclear power cost-competitive with utility-scale solar?", "tags": [ "disputed", "energy", "policy" ], "session_id": "s_06FZA3GMDREQ9T5C", "status": "done", "claims": 37, "scorecard": { "session_id": "s_06FZA3GMDREQ9T5C", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0168 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "49 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "37 of 37 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 21.62162162162162, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 37 claims (https://cleantechnica.com/2021/11/17/utility-scale-solar-reaches-lcoe-range-between-2-4%C2%A2-per-kwh-in-the-usa-record-low)" }, { "name": "cost per claim", "status": "measured", "value": 0.000455, "unit": "usd", "detail": "$0.0168 over 37 claims = $0.0004 each" }, { "name": "tool calls", "status": "measured", "value": 49, "unit": "calls", "detail": "llm 48 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "37 of 37 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 24.324324324324323, "unit": "%", "detail": "37 claim(s) render as 28 finding(s)" }, { "name": "disagreement rate", "status": "measured", "unit": "%", "detail": "0 contradiction(s) involving 0 of 37 claim(s)" }, { "name": "staleness separation", "status": "n/a", "unit": "edges", "detail": "no conflicting claims to separate" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "gpt4-mmlu", "question": "what score does GPT-4 achieve on MMLU?", "tags": [ "disputed", "ml-benchmarks", "numeric" ], "session_id": "s_06FZA40FC8AK7G7B", "status": "done", "claims": 28, "scorecard": { "session_id": "s_06FZA40FC8AK7G7B", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0156 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "47 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "28 of 28 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 28.571428571428573, "unit": "%", "detail": "5 distinct source(s); the largest contributed 8 of 28 claims (https://arxiv.org/html/2406.01574v2)" }, { "name": "cost per claim", "status": "measured", "value": 0.00056, "unit": "usd", "detail": "$0.0156 over 28 claims = $0.0005 each" }, { "name": "tool calls", "status": "measured", "value": 47, "unit": "calls", "detail": "llm 46 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "28 of 28 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 25, "unit": "%", "detail": "28 claim(s) render as 21 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 17.857142857142858, "unit": "%", "detail": "3 contradiction(s) involving 5 of 28 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 3 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "red-meat-mortality", "question": "does eating red meat increase all-cause mortality?", "tags": [ "disputed", "nutrition" ], "session_id": "s_06FZA4D33D0DXQ8C", "status": "done", "claims": 46, "scorecard": { "session_id": "s_06FZA4D33D0DXQ8C", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0149 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "65 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "46 of 46 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 17.391304347826086, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 46 claims (https://pmc.ncbi.nlm.nih.gov/articles/PMC3712342)" }, { "name": "cost per claim", "status": "measured", "value": 0.000324, "unit": "usd", "detail": "$0.0149 over 46 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 65, "unit": "calls", "detail": "fetch 6 · llm 58 · search 1 · 2 errored" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "46 of 46 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 17.391304347826086, "unit": "%", "detail": "46 claim(s) render as 38 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 8.695652173913043, "unit": "%", "detail": "2 contradiction(s) involving 4 of 46 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 2 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "measured", "value": 100, "unit": "%", "detail": "4 of 4 re-read claim(s) confirmed by their own source (1 inconclusive, excluded)" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "intermittent-fasting", "question": "is intermittent fasting more effective than continuous calorie restriction for weight loss?", "tags": [ "disputed", "nutrition" ], "session_id": "s_06FZA50VDY17YDH9", "status": "done", "claims": 47, "scorecard": { "session_id": "s_06FZA50VDY17YDH9", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0185 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "67 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "47 of 47 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 17.02127659574468, "unit": "%", "detail": "6 distinct source(s); the largest contributed 8 of 47 claims (https://link.springer.com/article/10.1186/s12967-018-1748-4)" }, { "name": "cost per claim", "status": "measured", "value": 0.000394, "unit": "usd", "detail": "$0.0185 over 47 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 67, "unit": "calls", "detail": "llm 66 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "47 of 47 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 25.53191489361702, "unit": "%", "detail": "47 claim(s) render as 35 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 6.382978723404255, "unit": "%", "detail": "2 contradiction(s) involving 3 of 47 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 2 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "predimed-cvd", "question": "does the Mediterranean diet reduce cardiovascular events?", "tags": [ "disputed", "stale", "nutrition" ], "session_id": "s_06FZA5N61FYFNKXE", "status": "done", "claims": 29, "scorecard": { "session_id": "s_06FZA5N61FYFNKXE", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0116 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "36 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "29 of 29 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 27.586206896551722, "unit": "%", "detail": "4 distinct source(s); the largest contributed 8 of 29 claims (https://pubmed.ncbi.nlm.nih.gov/38431146)" }, { "name": "cost per claim", "status": "measured", "value": 0.0004, "unit": "usd", "detail": "$0.0116 over 29 claims = $0.0004 each" }, { "name": "tool calls", "status": "measured", "value": 36, "unit": "calls", "detail": "llm 35 · search 1" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "29 of 29 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 34.48275862068966, "unit": "%", "detail": "29 claim(s) render as 19 finding(s)" }, { "name": "disagreement rate", "status": "measured", "unit": "%", "detail": "0 contradiction(s) involving 0 of 29 claim(s)" }, { "name": "staleness separation", "status": "n/a", "unit": "edges", "detail": "no conflicting claims to separate" }, { "name": "grounding rate", "status": "blocked", "reason": "no claim was re-read: §11.5.2's grounding pass needs escrow left over and a fetcher configured" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } }, { "id": "minimum-wage-employment", "question": "does raising the minimum wage reduce employment?", "tags": [ "disputed", "economics" ], "session_id": "s_06FZA611KY2VYASH", "status": "done", "claims": 130, "scorecard": { "session_id": "s_06FZA611KY2VYASH", "metrics": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "spent $0.0454 of $0.4000, no overshoot" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "165 call(s) reconcile" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "no reservations left held" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "130 of 130 claims well-formed" }, { "name": "source concentration", "status": "measured", "value": 10.76923076923077, "unit": "%", "detail": "17 distinct source(s); the largest contributed 14 of 130 claims (https://www.cbpp.org/research/economy/the-minimum-wage)" }, { "name": "cost per claim", "status": "measured", "value": 0.000349, "unit": "usd", "detail": "$0.0454 over 130 claims = $0.0003 each" }, { "name": "tool calls", "status": "measured", "value": 165, "unit": "calls", "detail": "fetch 3 · llm 159 · search 3 · 1 errored" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "130 of 130 claim(s) verified" }, { "name": "duplicate collapse", "status": "measured", "value": 17.692307692307693, "unit": "%", "detail": "130 claim(s) render as 107 finding(s)" }, { "name": "disagreement rate", "status": "measured", "value": 10.76923076923077, "unit": "%", "detail": "9 contradiction(s) involving 14 of 130 claim(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "0 of 9 conflict(s) resolved as staleness by publication date — check whether the sources carried publication dates at all" }, { "name": "grounding rate", "status": "measured", "value": 100, "unit": "%", "detail": "5 of 5 re-read claim(s) confirmed by their own source" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] } } ], "aggregate": [ { "name": "budget overshoot", "status": "measured", "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "ledger consistency", "status": "measured", "value": 1, "unit": "ok", "detail": "mean of 10 question(s)" }, { "name": "holds released", "status": "measured", "unit": "held", "detail": "mean of 10 question(s)" }, { "name": "claim integrity", "status": "measured", "value": 100, "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "source concentration", "status": "measured", "value": 20.683275267408735, "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "cost per claim", "status": "measured", "value": 0.0003974, "unit": "usd", "detail": "mean of 10 question(s)" }, { "name": "tool calls", "status": "measured", "value": 66.3, "unit": "calls", "detail": "mean of 10 question(s)" }, { "name": "verification coverage", "status": "measured", "value": 100, "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "duplicate collapse", "status": "measured", "value": 23.83481288042837, "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "disagreement rate", "status": "measured", "value": 11.252507334375975, "unit": "%", "detail": "mean of 10 question(s)" }, { "name": "staleness separation", "status": "measured", "unit": "edges", "detail": "mean of 8 question(s)" }, { "name": "grounding rate", "status": "measured", "value": 100, "unit": "%", "detail": "mean of 2 question(s); 8 blocked" }, { "name": "exfil regression", "status": "blocked", "reason": "this session crossed no local data; the assertion is enforced at the gate on every envelope (internal/compute/gate, §14.3) and is scored here only for a session that used a connector" }, { "name": "claim precision", "status": "blocked", "reason": "needs labelled answers; the question corpus (§14.2) is not built yet" }, { "name": "contradiction recall", "status": "blocked", "reason": "the Verifier now finds contradictions (see disagreement rate), but recall needs planted disagreements to measure what it MISSED — the question corpus (§14.2)" }, { "name": "staleness detection", "status": "blocked", "reason": "supersedes edges now exist (see staleness separation), but detection needs a corpus with known-stale sources to measure what was missed (§14.2)" }, { "name": "citation accuracy", "status": "blocked", "reason": "not requested; re-reads every cited source, so pass --citations (free and deterministic under MOLE_RECORD=replay)" } ] }