{ "kind": "ainglish.panel.items.v1", "proposal": "proxy-m-say-when-the-evidence-you-measured-is-a-proxy-for-th-2", "form": "X proxy()", "baseline": "complete_careful_english_mapping", "comparator": "careful", "scope_note": "Rosetta's pre-registered design in a two-arm harness: 30 proxy scenarios x 2 held-out probes (what did the writer directly check? / if the claim later fails, what was wrong?) with option vocabulary sharing no stem with proxy/verify/infer; the ainglish arm is `X proxy(); reading`, and the comparator is (careful) the mapping applied verbatim [CARRIER], (bare) 'X, and I measured M' [descriptive], (obs) `X obs(M); reading` [distinctiveness]. Calibration: English 'checked directly' vs Ainglish proxy(). Never pooled across comparators.", "real_items": 60, "calibration_items": 8, "sha256": "6dfd5e12b61df2b766c1ff05866fa4b2ff3db2360f7e87fc49afce9fbc2dc170", "items": [ { "id": "pr-car-01q1", "scenario": 26, "probe": 1, "english": "The code is tested; the evidence I directly verified is coverage (coverage of 91 percent); coverage is a proxy for whether the code is tested, not the same thing; the inference from coverage to whether the code is tested is the load-bearing step and it is unverified.", "ainglish": "The code is tested proxy(); coverage of 91 percent.", "question": "What did the writer directly check?", "options": [ "the measurement (coverage) only", "the claim itself (that the code is tested) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (coverage) only" }, { "id": "pr-car-01q2", "scenario": 26, "probe": 2, "english": "The code is tested; the evidence I directly verified is coverage (coverage of 91 percent); coverage is a proxy for whether the code is tested, not the same thing; the inference from coverage to whether the code is tested is the load-bearing step and it is unverified.", "ainglish": "The code is tested proxy(); coverage of 91 percent.", "question": "Suppose it later turns out that it is NOT the case that the code is tested. Going only by the sentence, what was wrong?", "options": [ "the step from coverage to the claim", "nothing — the claim was checked directly", "cannot tell", "the coverage reading itself" ], "answer": "the step from coverage to the claim" }, { "id": "pr-car-02q1", "scenario": 23, "probe": 1, "english": "The link is secure; the evidence I directly verified is padlock icon (the padlock icon showing); padlock icon is a proxy for whether the link is secure, not the same thing; the inference from padlock icon to whether the link is secure is the load-bearing step and it is unverified.", "ainglish": "The link is secure proxy(); the padlock icon showing.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the link is secure) only", "both the measurement and the claim", "cannot tell", "the measurement (padlock icon) only" ], "answer": "the measurement (padlock icon) only" }, { "id": "pr-car-02q2", "scenario": 23, "probe": 2, "english": "The link is secure; the evidence I directly verified is padlock icon (the padlock icon showing); padlock icon is a proxy for whether the link is secure, not the same thing; the inference from padlock icon to whether the link is secure is the load-bearing step and it is unverified.", "ainglish": "The link is secure proxy(); the padlock icon showing.", "question": "Suppose it later turns out that it is NOT the case that the link is secure. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the padlock icon reading itself", "the step from padlock icon to the claim" ], "answer": "the step from padlock icon to the claim" }, { "id": "pr-car-03q1", "scenario": 19, "probe": 1, "english": "The sensor is calibrated; the evidence I directly verified is self test (the self-test passing); self test is a proxy for whether the sensor is calibrated, not the same thing; the inference from self test to whether the sensor is calibrated is the load-bearing step and it is unverified.", "ainglish": "The sensor is calibrated proxy(); the self-test passing.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (self test) only", "the claim itself (that the sensor is calibrated) only" ], "answer": "the measurement (self test) only" }, { "id": "pr-car-03q2", "scenario": 19, "probe": 2, "english": "The sensor is calibrated; the evidence I directly verified is self test (the self-test passing); self test is a proxy for whether the sensor is calibrated, not the same thing; the inference from self test to whether the sensor is calibrated is the load-bearing step and it is unverified.", "ainglish": "The sensor is calibrated proxy(); the self-test passing.", "question": "Suppose it later turns out that it is NOT the case that the sensor is calibrated. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the self test reading itself", "the step from self test to the claim", "nothing — the claim was checked directly" ], "answer": "the step from self test to the claim" }, { "id": "pr-car-04q1", "scenario": 27, "probe": 1, "english": "The alert was seen; the evidence I directly verified is notification delivered (the notification marked delivered); notification delivered is a proxy for whether the alert was seen, not the same thing; the inference from notification delivered to whether the alert was seen is the load-bearing step and it is unverified.", "ainglish": "The alert was seen proxy(); the notification marked delivered.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (notification delivered) only", "the claim itself (that the alert was seen) only", "both the measurement and the claim" ], "answer": "the measurement (notification delivered) only" }, { "id": "pr-car-04q2", "scenario": 27, "probe": 2, "english": "The alert was seen; the evidence I directly verified is notification delivered (the notification marked delivered); notification delivered is a proxy for whether the alert was seen, not the same thing; the inference from notification delivered to whether the alert was seen is the load-bearing step and it is unverified.", "ainglish": "The alert was seen proxy(); the notification marked delivered.", "question": "Suppose it later turns out that it is NOT the case that the alert was seen. Going only by the sentence, what was wrong?", "options": [ "the notification delivered reading itself", "the step from notification delivered to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from notification delivered to the claim" }, { "id": "pr-car-05q1", "scenario": 9, "probe": 1, "english": "The cache is warm; the evidence I directly verified is hit rate (a 94 percent hit rate); hit rate is a proxy for whether the cache is warm, not the same thing; the inference from hit rate to whether the cache is warm is the load-bearing step and it is unverified.", "ainglish": "The cache is warm proxy(); a 94 percent hit rate.", "question": "What did the writer directly check?", "options": [ "the measurement (hit rate) only", "the claim itself (that the cache is warm) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (hit rate) only" }, { "id": "pr-car-05q2", "scenario": 9, "probe": 2, "english": "The cache is warm; the evidence I directly verified is hit rate (a 94 percent hit rate); hit rate is a proxy for whether the cache is warm, not the same thing; the inference from hit rate to whether the cache is warm is the load-bearing step and it is unverified.", "ainglish": "The cache is warm proxy(); a 94 percent hit rate.", "question": "Suppose it later turns out that it is NOT the case that the cache is warm. Going only by the sentence, what was wrong?", "options": [ "the step from hit rate to the claim", "nothing — the claim was checked directly", "cannot tell", "the hit rate reading itself" ], "answer": "the step from hit rate to the claim" }, { "id": "pr-car-06q1", "scenario": 21, "probe": 1, "english": "The dataset is deduplicated; the evidence I directly verified is hash uniqueness (all row hashes unique); hash uniqueness is a proxy for whether the dataset is deduplicated, not the same thing; the inference from hash uniqueness to whether the dataset is deduplicated is the load-bearing step and it is unverified.", "ainglish": "The dataset is deduplicated proxy(); all row hashes unique.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the dataset is deduplicated) only", "both the measurement and the claim", "cannot tell", "the measurement (hash uniqueness) only" ], "answer": "the measurement (hash uniqueness) only" }, { "id": "pr-car-06q2", "scenario": 21, "probe": 2, "english": "The dataset is deduplicated; the evidence I directly verified is hash uniqueness (all row hashes unique); hash uniqueness is a proxy for whether the dataset is deduplicated, not the same thing; the inference from hash uniqueness to whether the dataset is deduplicated is the load-bearing step and it is unverified.", "ainglish": "The dataset is deduplicated proxy(); all row hashes unique.", "question": "Suppose it later turns out that it is NOT the case that the dataset is deduplicated. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the hash uniqueness reading itself", "the step from hash uniqueness to the claim" ], "answer": "the step from hash uniqueness to the claim" }, { "id": "pr-car-07q1", "scenario": 14, "probe": 1, "english": "The key was rotated; the evidence I directly verified is timestamp change (a new mtime on the key file); timestamp change is a proxy for whether the key was rotated, not the same thing; the inference from timestamp change to whether the key was rotated is the load-bearing step and it is unverified.", "ainglish": "The key was rotated proxy(); a new mtime on the key file.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (timestamp change) only", "the claim itself (that the key was rotated) only" ], "answer": "the measurement (timestamp change) only" }, { "id": "pr-car-07q2", "scenario": 14, "probe": 2, "english": "The key was rotated; the evidence I directly verified is timestamp change (a new mtime on the key file); timestamp change is a proxy for whether the key was rotated, not the same thing; the inference from timestamp change to whether the key was rotated is the load-bearing step and it is unverified.", "ainglish": "The key was rotated proxy(); a new mtime on the key file.", "question": "Suppose it later turns out that it is NOT the case that the key was rotated. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the timestamp change reading itself", "the step from timestamp change to the claim", "nothing — the claim was checked directly" ], "answer": "the step from timestamp change to the claim" }, { "id": "pr-car-08q1", "scenario": 16, "probe": 1, "english": "The translation is faithful; the evidence I directly verified is back translation (back-translation matching 88 percent); back translation is a proxy for whether the translation is faithful, not the same thing; the inference from back translation to whether the translation is faithful is the load-bearing step and it is unverified.", "ainglish": "The translation is faithful proxy(); back-translation matching 88 percent.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (back translation) only", "the claim itself (that the translation is faithful) only", "both the measurement and the claim" ], "answer": "the measurement (back translation) only" }, { "id": "pr-car-08q2", "scenario": 16, "probe": 2, "english": "The translation is faithful; the evidence I directly verified is back translation (back-translation matching 88 percent); back translation is a proxy for whether the translation is faithful, not the same thing; the inference from back translation to whether the translation is faithful is the load-bearing step and it is unverified.", "ainglish": "The translation is faithful proxy(); back-translation matching 88 percent.", "question": "Suppose it later turns out that it is NOT the case that the translation is faithful. Going only by the sentence, what was wrong?", "options": [ "the back translation reading itself", "the step from back translation to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from back translation to the claim" }, { "id": "pr-car-09q1", "scenario": 15, "probe": 1, "english": "The queue is drained; the evidence I directly verified is queue length (queue length 0 at 09:00); queue length is a proxy for whether the queue is drained, not the same thing; the inference from queue length to whether the queue is drained is the load-bearing step and it is unverified.", "ainglish": "The queue is drained proxy(); queue length 0 at 09:00.", "question": "What did the writer directly check?", "options": [ "the measurement (queue length) only", "the claim itself (that the queue is drained) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (queue length) only" }, { "id": "pr-car-09q2", "scenario": 15, "probe": 2, "english": "The queue is drained; the evidence I directly verified is queue length (queue length 0 at 09:00); queue length is a proxy for whether the queue is drained, not the same thing; the inference from queue length to whether the queue is drained is the load-bearing step and it is unverified.", "ainglish": "The queue is drained proxy(); queue length 0 at 09:00.", "question": "Suppose it later turns out that it is NOT the case that the queue is drained. Going only by the sentence, what was wrong?", "options": [ "the step from queue length to the claim", "nothing — the claim was checked directly", "cannot tell", "the queue length reading itself" ], "answer": "the step from queue length to the claim" }, { "id": "pr-car-10q1", "scenario": 4, "probe": 1, "english": "The backup is restorable; the evidence I directly verified is file count (all 4,120 files present); file count is a proxy for whether the backup is restorable, not the same thing; the inference from file count to whether the backup is restorable is the load-bearing step and it is unverified.", "ainglish": "The backup is restorable proxy(); all 4,120 files present.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the backup is restorable) only", "both the measurement and the claim", "cannot tell", "the measurement (file count) only" ], "answer": "the measurement (file count) only" }, { "id": "pr-car-10q2", "scenario": 4, "probe": 2, "english": "The backup is restorable; the evidence I directly verified is file count (all 4,120 files present); file count is a proxy for whether the backup is restorable, not the same thing; the inference from file count to whether the backup is restorable is the load-bearing step and it is unverified.", "ainglish": "The backup is restorable proxy(); all 4,120 files present.", "question": "Suppose it later turns out that it is NOT the case that the backup is restorable. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the file count reading itself", "the step from file count to the claim" ], "answer": "the step from file count to the claim" }, { "id": "pr-car-11q1", "scenario": 7, "probe": 1, "english": "The patch fixed the bug; the evidence I directly verified is test pass (the regression test passing); test pass is a proxy for whether the patch fixed the bug, not the same thing; the inference from test pass to whether the patch fixed the bug is the load-bearing step and it is unverified.", "ainglish": "The patch fixed the bug proxy(); the regression test passing.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (test pass) only", "the claim itself (that the patch fixed the bug) only" ], "answer": "the measurement (test pass) only" }, { "id": "pr-car-11q2", "scenario": 7, "probe": 2, "english": "The patch fixed the bug; the evidence I directly verified is test pass (the regression test passing); test pass is a proxy for whether the patch fixed the bug, not the same thing; the inference from test pass to whether the patch fixed the bug is the load-bearing step and it is unverified.", "ainglish": "The patch fixed the bug proxy(); the regression test passing.", "question": "Suppose it later turns out that it is NOT the case that the patch fixed the bug. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the test pass reading itself", "the step from test pass to the claim", "nothing — the claim was checked directly" ], "answer": "the step from test pass to the claim" }, { "id": "pr-car-12q1", "scenario": 22, "probe": 1, "english": "The customer is satisfied; the evidence I directly verified is nps score (an NPS of 9); nps score is a proxy for whether the customer is satisfied, not the same thing; the inference from nps score to whether the customer is satisfied is the load-bearing step and it is unverified.", "ainglish": "The customer is satisfied proxy(); an NPS of 9.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (nps score) only", "the claim itself (that the customer is satisfied) only", "both the measurement and the claim" ], "answer": "the measurement (nps score) only" }, { "id": "pr-car-12q2", "scenario": 22, "probe": 2, "english": "The customer is satisfied; the evidence I directly verified is nps score (an NPS of 9); nps score is a proxy for whether the customer is satisfied, not the same thing; the inference from nps score to whether the customer is satisfied is the load-bearing step and it is unverified.", "ainglish": "The customer is satisfied proxy(); an NPS of 9.", "question": "Suppose it later turns out that it is NOT the case that the customer is satisfied. Going only by the sentence, what was wrong?", "options": [ "the nps score reading itself", "the step from nps score to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from nps score to the claim" }, { "id": "pr-car-13q1", "scenario": 17, "probe": 1, "english": "The anchor is confirmed; the evidence I directly verified is calendar ack (acknowledgements from 4 calendars); calendar ack is a proxy for whether the anchor is confirmed, not the same thing; the inference from calendar ack to whether the anchor is confirmed is the load-bearing step and it is unverified.", "ainglish": "The anchor is confirmed proxy(); acknowledgements from 4 calendars.", "question": "What did the writer directly check?", "options": [ "the measurement (calendar ack) only", "the claim itself (that the anchor is confirmed) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (calendar ack) only" }, { "id": "pr-car-13q2", "scenario": 17, "probe": 2, "english": "The anchor is confirmed; the evidence I directly verified is calendar ack (acknowledgements from 4 calendars); calendar ack is a proxy for whether the anchor is confirmed, not the same thing; the inference from calendar ack to whether the anchor is confirmed is the load-bearing step and it is unverified.", "ainglish": "The anchor is confirmed proxy(); acknowledgements from 4 calendars.", "question": "Suppose it later turns out that it is NOT the case that the anchor is confirmed. Going only by the sentence, what was wrong?", "options": [ "the step from calendar ack to the claim", "nothing — the claim was checked directly", "cannot tell", "the calendar ack reading itself" ], "answer": "the step from calendar ack to the claim" }, { "id": "pr-car-14q1", "scenario": 18, "probe": 1, "english": "The contributor accepted the terms; the evidence I directly verified is checkbox (the checkbox ticked); checkbox is a proxy for whether the contributor accepted the terms, not the same thing; the inference from checkbox to whether the contributor accepted the terms is the load-bearing step and it is unverified.", "ainglish": "The contributor accepted the terms proxy(); the checkbox ticked.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the contributor accepted the terms) only", "both the measurement and the claim", "cannot tell", "the measurement (checkbox) only" ], "answer": "the measurement (checkbox) only" }, { "id": "pr-car-14q2", "scenario": 18, "probe": 2, "english": "The contributor accepted the terms; the evidence I directly verified is checkbox (the checkbox ticked); checkbox is a proxy for whether the contributor accepted the terms, not the same thing; the inference from checkbox to whether the contributor accepted the terms is the load-bearing step and it is unverified.", "ainglish": "The contributor accepted the terms proxy(); the checkbox ticked.", "question": "Suppose it later turns out that it is NOT the case that the contributor accepted the terms. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the checkbox reading itself", "the step from checkbox to the claim" ], "answer": "the step from checkbox to the claim" }, { "id": "pr-car-15q1", "scenario": 12, "probe": 1, "english": "The deploy reached every region; the evidence I directly verified is dns propagation (DNS answers from all 6 resolvers); dns propagation is a proxy for whether the deploy reached every region, not the same thing; the inference from dns propagation to whether the deploy reached every region is the load-bearing step and it is unverified.", "ainglish": "The deploy reached every region proxy(); DNS answers from all 6 resolvers.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (dns propagation) only", "the claim itself (that the deploy reached every region) only" ], "answer": "the measurement (dns propagation) only" }, { "id": "pr-car-15q2", "scenario": 12, "probe": 2, "english": "The deploy reached every region; the evidence I directly verified is dns propagation (DNS answers from all 6 resolvers); dns propagation is a proxy for whether the deploy reached every region, not the same thing; the inference from dns propagation to whether the deploy reached every region is the load-bearing step and it is unverified.", "ainglish": "The deploy reached every region proxy(); DNS answers from all 6 resolvers.", "question": "Suppose it later turns out that it is NOT the case that the deploy reached every region. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the dns propagation reading itself", "the step from dns propagation to the claim", "nothing — the claim was checked directly" ], "answer": "the step from dns propagation to the claim" }, { "id": "pr-car-16q1", "scenario": 11, "probe": 1, "english": "The document is accurate; the evidence I directly verified is spell check (zero spelling errors); spell check is a proxy for whether the document is accurate, not the same thing; the inference from spell check to whether the document is accurate is the load-bearing step and it is unverified.", "ainglish": "The document is accurate proxy(); zero spelling errors.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (spell check) only", "the claim itself (that the document is accurate) only", "both the measurement and the claim" ], "answer": "the measurement (spell check) only" }, { "id": "pr-car-16q2", "scenario": 11, "probe": 2, "english": "The document is accurate; the evidence I directly verified is spell check (zero spelling errors); spell check is a proxy for whether the document is accurate, not the same thing; the inference from spell check to whether the document is accurate is the load-bearing step and it is unverified.", "ainglish": "The document is accurate proxy(); zero spelling errors.", "question": "Suppose it later turns out that it is NOT the case that the document is accurate. Going only by the sentence, what was wrong?", "options": [ "the spell check reading itself", "the step from spell check to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from spell check to the claim" }, { "id": "pr-car-17q1", "scenario": 1, "probe": 1, "english": "The service is healthy; the evidence I directly verified is ping latency (ping under 5 ms); ping latency is a proxy for whether the service is healthy, not the same thing; the inference from ping latency to whether the service is healthy is the load-bearing step and it is unverified.", "ainglish": "The service is healthy proxy(); ping under 5 ms.", "question": "What did the writer directly check?", "options": [ "the measurement (ping latency) only", "the claim itself (that the service is healthy) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (ping latency) only" }, { "id": "pr-car-17q2", "scenario": 1, "probe": 2, "english": "The service is healthy; the evidence I directly verified is ping latency (ping under 5 ms); ping latency is a proxy for whether the service is healthy, not the same thing; the inference from ping latency to whether the service is healthy is the load-bearing step and it is unverified.", "ainglish": "The service is healthy proxy(); ping under 5 ms.", "question": "Suppose it later turns out that it is NOT the case that the service is healthy. Going only by the sentence, what was wrong?", "options": [ "the step from ping latency to the claim", "nothing — the claim was checked directly", "cannot tell", "the ping latency reading itself" ], "answer": "the step from ping latency to the claim" }, { "id": "pr-car-18q1", "scenario": 3, "probe": 1, "english": "Users understood the notice; the evidence I directly verified is click through (click-through of 62 percent); click through is a proxy for whether users understood the notice, not the same thing; the inference from click through to whether users understood the notice is the load-bearing step and it is unverified.", "ainglish": "Users understood the notice proxy(); click-through of 62 percent.", "question": "What did the writer directly check?", "options": [ "the claim itself (that users understood the notice) only", "both the measurement and the claim", "cannot tell", "the measurement (click through) only" ], "answer": "the measurement (click through) only" }, { "id": "pr-car-18q2", "scenario": 3, "probe": 2, "english": "Users understood the notice; the evidence I directly verified is click through (click-through of 62 percent); click through is a proxy for whether users understood the notice, not the same thing; the inference from click through to whether users understood the notice is the load-bearing step and it is unverified.", "ainglish": "Users understood the notice proxy(); click-through of 62 percent.", "question": "Suppose it later turns out that it is NOT the case that users understood the notice. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the click through reading itself", "the step from click through to the claim" ], "answer": "the step from click through to the claim" }, { "id": "pr-car-19q1", "scenario": 24, "probe": 1, "english": "The disk is healthy; the evidence I directly verified is smart status (SMART status OK); smart status is a proxy for whether the disk is healthy, not the same thing; the inference from smart status to whether the disk is healthy is the load-bearing step and it is unverified.", "ainglish": "The disk is healthy proxy(); SMART status OK.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (smart status) only", "the claim itself (that the disk is healthy) only" ], "answer": "the measurement (smart status) only" }, { "id": "pr-car-19q2", "scenario": 24, "probe": 2, "english": "The disk is healthy; the evidence I directly verified is smart status (SMART status OK); smart status is a proxy for whether the disk is healthy, not the same thing; the inference from smart status to whether the disk is healthy is the load-bearing step and it is unverified.", "ainglish": "The disk is healthy proxy(); SMART status OK.", "question": "Suppose it later turns out that it is NOT the case that the disk is healthy. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the smart status reading itself", "the step from smart status to the claim", "nothing — the claim was checked directly" ], "answer": "the step from smart status to the claim" }, { "id": "pr-car-20q1", "scenario": 28, "probe": 1, "english": "The argument is sound; the evidence I directly verified is citation count (14 citations); citation count is a proxy for whether the argument is sound, not the same thing; the inference from citation count to whether the argument is sound is the load-bearing step and it is unverified.", "ainglish": "The argument is sound proxy(); 14 citations.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (citation count) only", "the claim itself (that the argument is sound) only", "both the measurement and the claim" ], "answer": "the measurement (citation count) only" }, { "id": "pr-car-20q2", "scenario": 28, "probe": 2, "english": "The argument is sound; the evidence I directly verified is citation count (14 citations); citation count is a proxy for whether the argument is sound, not the same thing; the inference from citation count to whether the argument is sound is the load-bearing step and it is unverified.", "ainglish": "The argument is sound proxy(); 14 citations.", "question": "Suppose it later turns out that it is NOT the case that the argument is sound. Going only by the sentence, what was wrong?", "options": [ "the citation count reading itself", "the step from citation count to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from citation count to the claim" }, { "id": "pr-car-21q1", "scenario": 6, "probe": 1, "english": "The vote was representative; the evidence I directly verified is turnout (turnout of 71 percent); turnout is a proxy for whether the vote was representative, not the same thing; the inference from turnout to whether the vote was representative is the load-bearing step and it is unverified.", "ainglish": "The vote was representative proxy(); turnout of 71 percent.", "question": "What did the writer directly check?", "options": [ "the measurement (turnout) only", "the claim itself (that the vote was representative) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (turnout) only" }, { "id": "pr-car-21q2", "scenario": 6, "probe": 2, "english": "The vote was representative; the evidence I directly verified is turnout (turnout of 71 percent); turnout is a proxy for whether the vote was representative, not the same thing; the inference from turnout to whether the vote was representative is the load-bearing step and it is unverified.", "ainglish": "The vote was representative proxy(); turnout of 71 percent.", "question": "Suppose it later turns out that it is NOT the case that the vote was representative. Going only by the sentence, what was wrong?", "options": [ "the step from turnout to the claim", "nothing — the claim was checked directly", "cannot tell", "the turnout reading itself" ], "answer": "the step from turnout to the claim" }, { "id": "pr-car-22q1", "scenario": 0, "probe": 1, "english": "9 people read the message; the evidence I directly verified is page fetches (9 of 9 fetches); page fetches is a proxy for whether 9 people read the message, not the same thing; the inference from page fetches to whether 9 people read the message is the load-bearing step and it is unverified.", "ainglish": "9 people read the message proxy(); 9 of 9 fetches.", "question": "What did the writer directly check?", "options": [ "the claim itself (that 9 people read the message) only", "both the measurement and the claim", "cannot tell", "the measurement (page fetches) only" ], "answer": "the measurement (page fetches) only" }, { "id": "pr-car-22q2", "scenario": 0, "probe": 2, "english": "9 people read the message; the evidence I directly verified is page fetches (9 of 9 fetches); page fetches is a proxy for whether 9 people read the message, not the same thing; the inference from page fetches to whether 9 people read the message is the load-bearing step and it is unverified.", "ainglish": "9 people read the message proxy(); 9 of 9 fetches.", "question": "Suppose it later turns out that it is NOT the case that 9 people read the message. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the page fetches reading itself", "the step from page fetches to the claim" ], "answer": "the step from page fetches to the claim" }, { "id": "pr-car-23q1", "scenario": 29, "probe": 1, "english": "The API is compatible; the evidence I directly verified is schema diff (an empty schema diff); schema diff is a proxy for whether the API is compatible, not the same thing; the inference from schema diff to whether the API is compatible is the load-bearing step and it is unverified.", "ainglish": "The API is compatible proxy(); an empty schema diff.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (schema diff) only", "the claim itself (that the API is compatible) only" ], "answer": "the measurement (schema diff) only" }, { "id": "pr-car-23q2", "scenario": 29, "probe": 2, "english": "The API is compatible; the evidence I directly verified is schema diff (an empty schema diff); schema diff is a proxy for whether the API is compatible, not the same thing; the inference from schema diff to whether the API is compatible is the load-bearing step and it is unverified.", "ainglish": "The API is compatible proxy(); an empty schema diff.", "question": "Suppose it later turns out that it is NOT the case that the API is compatible. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the schema diff reading itself", "the step from schema diff to the claim", "nothing — the claim was checked directly" ], "answer": "the step from schema diff to the claim" }, { "id": "pr-car-24q1", "scenario": 13, "probe": 1, "english": "The audience was engaged; the evidence I directly verified is dwell time (median dwell of 3 minutes); dwell time is a proxy for whether the audience was engaged, not the same thing; the inference from dwell time to whether the audience was engaged is the load-bearing step and it is unverified.", "ainglish": "The audience was engaged proxy(); median dwell of 3 minutes.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (dwell time) only", "the claim itself (that the audience was engaged) only", "both the measurement and the claim" ], "answer": "the measurement (dwell time) only" }, { "id": "pr-car-24q2", "scenario": 13, "probe": 2, "english": "The audience was engaged; the evidence I directly verified is dwell time (median dwell of 3 minutes); dwell time is a proxy for whether the audience was engaged, not the same thing; the inference from dwell time to whether the audience was engaged is the load-bearing step and it is unverified.", "ainglish": "The audience was engaged proxy(); median dwell of 3 minutes.", "question": "Suppose it later turns out that it is NOT the case that the audience was engaged. Going only by the sentence, what was wrong?", "options": [ "the dwell time reading itself", "the step from dwell time to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from dwell time to the claim" }, { "id": "pr-car-25q1", "scenario": 10, "probe": 1, "english": "The agent is idle; the evidence I directly verified is cpu load (CPU load under 2 percent); cpu load is a proxy for whether the agent is idle, not the same thing; the inference from cpu load to whether the agent is idle is the load-bearing step and it is unverified.", "ainglish": "The agent is idle proxy(); CPU load under 2 percent.", "question": "What did the writer directly check?", "options": [ "the measurement (cpu load) only", "the claim itself (that the agent is idle) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (cpu load) only" }, { "id": "pr-car-25q2", "scenario": 10, "probe": 2, "english": "The agent is idle; the evidence I directly verified is cpu load (CPU load under 2 percent); cpu load is a proxy for whether the agent is idle, not the same thing; the inference from cpu load to whether the agent is idle is the load-bearing step and it is unverified.", "ainglish": "The agent is idle proxy(); CPU load under 2 percent.", "question": "Suppose it later turns out that it is NOT the case that the agent is idle. Going only by the sentence, what was wrong?", "options": [ "the step from cpu load to the claim", "nothing — the claim was checked directly", "cannot tell", "the cpu load reading itself" ], "answer": "the step from cpu load to the claim" }, { "id": "pr-car-26q1", "scenario": 5, "probe": 1, "english": "The model improved; the evidence I directly verified is loss curve (training loss down 12 percent); loss curve is a proxy for whether the model improved, not the same thing; the inference from loss curve to whether the model improved is the load-bearing step and it is unverified.", "ainglish": "The model improved proxy(); training loss down 12 percent.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the model improved) only", "both the measurement and the claim", "cannot tell", "the measurement (loss curve) only" ], "answer": "the measurement (loss curve) only" }, { "id": "pr-car-26q2", "scenario": 5, "probe": 2, "english": "The model improved; the evidence I directly verified is loss curve (training loss down 12 percent); loss curve is a proxy for whether the model improved, not the same thing; the inference from loss curve to whether the model improved is the load-bearing step and it is unverified.", "ainglish": "The model improved proxy(); training loss down 12 percent.", "question": "Suppose it later turns out that it is NOT the case that the model improved. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the loss curve reading itself", "the step from loss curve to the claim" ], "answer": "the step from loss curve to the claim" }, { "id": "pr-car-27q1", "scenario": 25, "probe": 1, "english": "The meeting happened; the evidence I directly verified is calendar entry (the calendar entry marked done); calendar entry is a proxy for whether the meeting happened, not the same thing; the inference from calendar entry to whether the meeting happened is the load-bearing step and it is unverified.", "ainglish": "The meeting happened proxy(); the calendar entry marked done.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (calendar entry) only", "the claim itself (that the meeting happened) only" ], "answer": "the measurement (calendar entry) only" }, { "id": "pr-car-27q2", "scenario": 25, "probe": 2, "english": "The meeting happened; the evidence I directly verified is calendar entry (the calendar entry marked done); calendar entry is a proxy for whether the meeting happened, not the same thing; the inference from calendar entry to whether the meeting happened is the load-bearing step and it is unverified.", "ainglish": "The meeting happened proxy(); the calendar entry marked done.", "question": "Suppose it later turns out that it is NOT the case that the meeting happened. Going only by the sentence, what was wrong?", "options": [ "cannot tell", "the calendar entry reading itself", "the step from calendar entry to the claim", "nothing — the claim was checked directly" ], "answer": "the step from calendar entry to the claim" }, { "id": "pr-car-28q1", "scenario": 20, "probe": 1, "english": "The room is empty; the evidence I directly verified is motion sensor (no motion for 10 minutes); motion sensor is a proxy for whether the room is empty, not the same thing; the inference from motion sensor to whether the room is empty is the load-bearing step and it is unverified.", "ainglish": "The room is empty proxy(); no motion for 10 minutes.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (motion sensor) only", "the claim itself (that the room is empty) only", "both the measurement and the claim" ], "answer": "the measurement (motion sensor) only" }, { "id": "pr-car-28q2", "scenario": 20, "probe": 2, "english": "The room is empty; the evidence I directly verified is motion sensor (no motion for 10 minutes); motion sensor is a proxy for whether the room is empty, not the same thing; the inference from motion sensor to whether the room is empty is the load-bearing step and it is unverified.", "ainglish": "The room is empty proxy(); no motion for 10 minutes.", "question": "Suppose it later turns out that it is NOT the case that the room is empty. Going only by the sentence, what was wrong?", "options": [ "the motion sensor reading itself", "the step from motion sensor to the claim", "nothing — the claim was checked directly", "cannot tell" ], "answer": "the step from motion sensor to the claim" }, { "id": "pr-car-29q1", "scenario": 8, "probe": 1, "english": "The reviewer read the diff; the evidence I directly verified is time open (the tab open for 40 minutes); time open is a proxy for whether the reviewer read the diff, not the same thing; the inference from time open to whether the reviewer read the diff is the load-bearing step and it is unverified.", "ainglish": "The reviewer read the diff proxy(); the tab open for 40 minutes.", "question": "What did the writer directly check?", "options": [ "the measurement (time open) only", "the claim itself (that the reviewer read the diff) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (time open) only" }, { "id": "pr-car-29q2", "scenario": 8, "probe": 2, "english": "The reviewer read the diff; the evidence I directly verified is time open (the tab open for 40 minutes); time open is a proxy for whether the reviewer read the diff, not the same thing; the inference from time open to whether the reviewer read the diff is the load-bearing step and it is unverified.", "ainglish": "The reviewer read the diff proxy(); the tab open for 40 minutes.", "question": "Suppose it later turns out that it is NOT the case that the reviewer read the diff. Going only by the sentence, what was wrong?", "options": [ "the step from time open to the claim", "nothing — the claim was checked directly", "cannot tell", "the time open reading itself" ], "answer": "the step from time open to the claim" }, { "id": "pr-car-30q1", "scenario": 2, "probe": 1, "english": "The migration succeeded; the evidence I directly verified is exit code (exit code 0); exit code is a proxy for whether the migration succeeded, not the same thing; the inference from exit code to whether the migration succeeded is the load-bearing step and it is unverified.", "ainglish": "The migration succeeded proxy(); exit code 0.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the migration succeeded) only", "both the measurement and the claim", "cannot tell", "the measurement (exit code) only" ], "answer": "the measurement (exit code) only" }, { "id": "pr-car-30q2", "scenario": 2, "probe": 2, "english": "The migration succeeded; the evidence I directly verified is exit code (exit code 0); exit code is a proxy for whether the migration succeeded, not the same thing; the inference from exit code to whether the migration succeeded is the load-bearing step and it is unverified.", "ainglish": "The migration succeeded proxy(); exit code 0.", "question": "Suppose it later turns out that it is NOT the case that the migration succeeded. Going only by the sentence, what was wrong?", "options": [ "nothing — the claim was checked directly", "cannot tell", "the exit code reading itself", "the step from exit code to the claim" ], "answer": "the step from exit code to the claim" }, { "id": "pr-car-cal-01", "calibration": true, "english": "9 people read the message; I checked that directly.", "ainglish": "9 people read the message proxy(): 9 of 9 fetches.", "question": "What did the writer directly check?", "options": [ "the measurement (page fetches) only", "the claim itself (that 9 people read the message) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (page fetches) only" }, { "id": "pr-car-cal-02", "calibration": true, "english": "The patch fixed the bug; I checked that directly.", "ainglish": "The patch fixed the bug proxy(): the regression test passing.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the patch fixed the bug) only", "both the measurement and the claim", "cannot tell", "the measurement (test pass) only" ], "answer": "the measurement (test pass) only" }, { "id": "pr-car-cal-03", "calibration": true, "english": "The key was rotated; I checked that directly.", "ainglish": "The key was rotated proxy(): a new mtime on the key file.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (timestamp change) only", "the claim itself (that the key was rotated) only" ], "answer": "the measurement (timestamp change) only" }, { "id": "pr-car-cal-04", "calibration": true, "english": "The dataset is deduplicated; I checked that directly.", "ainglish": "The dataset is deduplicated proxy(): all row hashes unique.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (hash uniqueness) only", "the claim itself (that the dataset is deduplicated) only", "both the measurement and the claim" ], "answer": "the measurement (hash uniqueness) only" }, { "id": "pr-car-cal-05", "calibration": true, "english": "The argument is sound; I checked that directly.", "ainglish": "The argument is sound proxy(): 14 citations.", "question": "What did the writer directly check?", "options": [ "the measurement (citation count) only", "the claim itself (that the argument is sound) only", "both the measurement and the claim", "cannot tell" ], "answer": "the measurement (citation count) only" }, { "id": "pr-car-cal-06", "calibration": true, "english": "The model improved; I checked that directly.", "ainglish": "The model improved proxy(): training loss down 12 percent.", "question": "What did the writer directly check?", "options": [ "the claim itself (that the model improved) only", "both the measurement and the claim", "cannot tell", "the measurement (loss curve) only" ], "answer": "the measurement (loss curve) only" }, { "id": "pr-car-cal-07", "calibration": true, "english": "The deploy reached every region; I checked that directly.", "ainglish": "The deploy reached every region proxy(): DNS answers from all 6 resolvers.", "question": "What did the writer directly check?", "options": [ "both the measurement and the claim", "cannot tell", "the measurement (dns propagation) only", "the claim itself (that the deploy reached every region) only" ], "answer": "the measurement (dns propagation) only" }, { "id": "pr-car-cal-08", "calibration": true, "english": "The sensor is calibrated; I checked that directly.", "ainglish": "The sensor is calibrated proxy(): the self-test passing.", "question": "What did the writer directly check?", "options": [ "cannot tell", "the measurement (self test) only", "the claim itself (that the sensor is calibrated) only", "both the measurement and the claim" ], "answer": "the measurement (self test) only" } ] }