Repository object · research-note
Source Records
Accepted research note in the public catalog.
- Media type
application/json- Object ID
em:research-note:sha256:1b39967c771de6e5874fb738847b6d42adaacca39f37b7074f2e00866b1b5041- Content digest
385bbc193be495597ef0fe2a675220ca443297e673b32947c8d8ffcd1a6366da
Also filed under
Source content
{
"schema": "https://epistemedia.org/research/source-records-v1.json",
"task_id": "EM-0032",
"evidence_cutoff": "2026-08-27",
"target_question": "How did a historical simulated UBE score reported for GPT-4 become a roughly 90th-percentile claim, and how does the rank change when the comparison population changes?",
"capture_method": "Credential-free GETs of public primary or authoritative carriers. Restricted works are represented by metadata and quote-minimal attributed spans; downloaded source bodies remain outside Git.",
"core_source_ids": [
"source-openai-v1",
"source-openai-v6",
"source-katz-vor",
"source-katz-ssrn",
"source-katz-git-snapshot",
"source-katz-figshare",
"source-illinois-feb-2018",
"source-illinois-jul-2018",
"source-illinois-feb-2019",
"source-ncbe-ube-mechanics",
"source-ncbe-mbe-2022",
"source-ncbe-snapshot-2022",
"source-ncbe-first-repeat-2022",
"source-martinez-vor",
"source-martinez-osf"
],
"sources": [
{
"source_id": "source-openai-v1",
"work_id": "work-openai-gpt4-report",
"edition_id": "edition-openai-2303.08774v1",
"core": true,
"title": "GPT-4 Technical Report",
"authors_or_org": "OpenAI",
"identifier": "arXiv:2303.08774v1; DOI 10.48550/arXiv.2303.08774",
"url": "https://arxiv.org/pdf/2303.08774v1",
"published": "2023-03-15",
"retrieved_at": "2026-08-27T04:45:06Z",
"media_type": "application/pdf",
"captured_bytes": 5229731,
"captured_sha256": "053056a10114d22e4c47b6b5be25e54c320b5f1beeae7466e8638dac0f5f5f66",
"license": "arXiv non-exclusive distribution license; no Creative Commons license confirmed",
"treatment": "link and quote minimally; do not redistribute the PDF",
"role": "launch-edition claim carrier",
"spans": [
{
"span_id": "span-openai-v1-abstract-top-ten",
"locator": "abstract, PDF p. 1",
"quote": "passing a simulated bar exam with a score around the top 10% of test takers",
"supports": "The launch report used the top-10-percent wording and said test takers, not lawyers."
},
{
"span_id": "span-openai-v1-table-score",
"locator": "Table 1, PDF p. 5",
"quote": "Uniform Bar Exam (MBE+MEE+MPT) 298 / 400 (~90th)",
"supports": "The report displayed 298/400 and approximately 90th percentile together."
},
{
"span_id": "span-openai-v1-scoring",
"locator": "Appendix A.5, PDF p. 25",
"quote": "Percentiles are based on the most recently available score distributions for test-takers of each exam type.",
"supports": "The report described a generic percentile method but did not name a UBE distribution in this passage."
},
{
"span_id": "span-openai-v1-snapshots",
"locator": "Appendix A.6, PDF p. 25",
"quote": "We ran GPT-4 multiple-choice questions using a model snapshot from March 1, 2023, whereas the free-response questions were run and scored using a non-final model snapshot from February 23, 2023.",
"supports": "The reported composite used two historical model snapshots rather than a stable current product identity."
},
{
"span_id": "span-openai-v1-collaborators",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "Appendix A.1, PDF p. 23",
"quote": "The Uniform Bar Exam was run by our collaborators at CaseText and Stanford CodeX.",
"supports": "The launch report declares author-social collaboration rather than an independent vendor-versus-study replication."
},
{
"span_id": "span-openai-v1-free-response-run",
"locator": "Appendix A.3, PDF p. 24",
"quote": "we simply ran these free response questions each only a single time at our best-guess temperature (0.6) and prompt",
"supports": "The free-response component was a single best-guess run, not a repeated performance estimate."
}
]
},
{
"source_id": "source-openai-v6",
"work_id": "work-openai-gpt4-report",
"edition_id": "edition-openai-2303.08774v6",
"core": true,
"title": "GPT-4 Technical Report",
"authors_or_org": "OpenAI",
"identifier": "arXiv:2303.08774v6; DOI 10.48550/arXiv.2303.08774",
"url": "https://arxiv.org/pdf/2303.08774v6",
"published": "2024-03-04",
"retrieved_at": "2026-08-27T04:45:07Z",
"media_type": "application/pdf",
"captured_bytes": 5245564,
"captured_sha256": "c33a66dadca2388d7b172d6293b00dc32b71110c6f38fafe0d41112e61be7774",
"license": "arXiv non-exclusive distribution license; no Creative Commons license confirmed",
"treatment": "link and quote minimally; do not redistribute the PDF",
"role": "later edition drift check",
"spans": [
{
"span_id": "span-openai-v6-table-score",
"locator": "Table 1, PDF p. 5",
"quote": "Uniform Bar Exam (MBE+MEE+MPT) 298 / 400 (~90th)",
"supports": "The later arXiv edition retained the same displayed score and percentile label."
}
]
},
{
"source_id": "source-katz-vor",
"work_id": "work-katz-gpt4-bar",
"edition_id": "edition-katz-rsta-2024",
"core": true,
"title": "GPT-4 passes the bar exam",
"authors_or_org": "Daniel Martin Katz, Michael James Bommarito, Shang Gao, and Pablo Arredondo",
"identifier": "DOI 10.1098/rsta.2023.0254; PMCID PMC10894685; PMID 38403056",
"url": "https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10894685/fullTextXML",
"published": "2024-02-26",
"retrieved_at": "2026-08-27T04:45:08Z",
"media_type": "application/xml",
"captured_bytes": 117084,
"captured_sha256": "d5a7b3d5cba67eb070f13e5e72700a11b2f081af664aad197acb37427aa47264",
"license": "CC BY 4.0",
"treatment": "attributed quotation permitted; keep excerpts minimal",
"role": "underlying study version of record",
"spans": [
{
"span_id": "span-katz-vor-abstract-score",
"locator": "abstract",
"quote": "Graded across the UBE components, in the manner in which a human test-taker would be, GPT-4 scores approximately 297 points",
"supports": "The study version of record reports approximately 297, not 298."
},
{
"span_id": "span-katz-vor-components",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "§4(d), Table 7",
"cells": [
{"cell_id": "cell-katz-gpt4-mbe", "row": "MBE", "column": "GPT-4", "text": "157 points"},
{"cell_id": "cell-katz-gpt4-mee", "row": "MEE", "column": "GPT-4", "text": "84 points"},
{"cell_id": "cell-katz-gpt4-mpt", "row": "MPT", "column": "GPT-4", "text": "56 points"},
{"cell_id": "cell-katz-gpt4-overall", "row": "overall score", "column": "GPT-4", "text": "297 points"}
],
"supports": "The version-of-record component scores sum to 297."
},
{
"span_id": "span-katz-vor-score-discrepancy",
"locator": "footnote 4",
"quote": "Best prompt and/or hyperparameter combination on the MBE would push this score to 298 or higher. Here, we report the MBE average of 75.7% which composites to a 297.",
"supports": "The source itself explains the 298-versus-approximately-297 discrepancy as a scoring-choice difference."
},
{
"span_id": "span-katz-vor-percentile-boundary",
"locator": "footnote 5",
"quote": "there is no publicly available July 2022 national bar exam percentiles against which to compare these results",
"supports": "The study states that the matching national July 2022 comparison distribution was unavailable."
},
{
"span_id": "span-katz-vor-range",
"locator": "footnote 5",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"quote": "While we are not fully convinced of the methodological approach taken in some subsequent analysis [78], we do agree that it would be better to consider the raw 297 UBE as falling within a range between 68th and 90th percentile (depending on the precise state and timing of the exam administration).",
"supports": "The version of record treats the percentile as comparison-population-dependent."
},
{
"span_id": "span-katz-vor-materials",
"format": "exact-segments",
"normalization": "collapse-whitespace",
"locator": "§3(a), materials",
"segments": [
{
"segment_id": "segment-katz-mee-mpt-materials",
"text": "For the MEE and the MPT, we collected the most recently released questions from the July 2022 Bar Examination."
},
{
"segment_id": "segment-katz-mbe-materials",
"text": "The MBE questions used in this study are official multistate bar examination questions from previous administrations of the UBE [65]."
}
],
"supports": "The study components share disclosed exam-item materials and administration lineage."
}
]
},
{
"source_id": "source-katz-ssrn",
"work_id": "work-katz-gpt4-bar",
"edition_id": "edition-katz-ssrn-4389233",
"core": true,
"title": "GPT-4 Passes the Bar Exam",
"authors_or_org": "Daniel Martin Katz, Michael James Bommarito, Shang Gao, and Pablo Arredondo",
"identifier": "DOI 10.2139/ssrn.4389233; SSRN 4389233",
"url": "https://api.crossref.org/works/10.2139/ssrn.4389233",
"published": "2023-03-15",
"retrieved_at": "2026-08-27T14:52:21Z",
"media_type": "application/json",
"captured_bytes": 16942,
"captured_sha256": "146a68d349dae70ece09fefe79cd04808a40afb8f1e19a4c63941678330cb90b",
"license": "no open license confirmed",
"treatment": "metadata only; do not redistribute the manuscript",
"role": "preprint identity; same study root",
"spans": []
},
{
"source_id": "source-katz-git-snapshot",
"work_id": "work-katz-gpt4-bar",
"edition_id": "edition-katz-git-commit-90997f7-tree-810bd4a",
"core": true,
"title": "gpt4-passes-the-bar repository snapshot",
"authors_or_org": "Michael James Bommarito and study authors",
"identifier": "Git commit 90997f740c7197f3f300b013e4345e2ad5621f96; resolved tree 810bd4a9a8ffb51e457715d2312d28d3e9657240",
"url": "https://api.github.com/repos/mjbommar/gpt4-passes-the-bar/git/trees/90997f740c7197f3f300b013e4345e2ad5621f96?recursive=1",
"published": "2023",
"retrieved_at": "2026-08-27T04:45:08Z",
"media_type": "application/json",
"captured_bytes": 29311,
"captured_sha256": "9541fd9b9677738aae5fac3048eaa5efc2a23ff5a97ecedec74492eb451e86fb",
"semantic_capture": {
"normalizer_id": "canonical-json-v1",
"command": ["python", "-m", "json.tool", "--sort-keys"],
"bytes": 23967,
"sha256": "77eed8a0e0bacf7ded2368209120bd7d2e1390419ece9421373bc9c696f83105"
},
"license": "no repository license at the pinned tree",
"treatment": "metadata inventory and quote-minimal readback only",
"role": "mechanical artifact manifestation; same study root",
"spans": []
},
{
"source_id": "source-katz-figshare",
"work_id": "work-katz-gpt4-bar",
"edition_id": "edition-katz-figshare-25018513-v1",
"core": true,
"title": "Appendix for GPT-4 passes the Bar Exam",
"authors_or_org": "Katz et al. / The Royal Society",
"identifier": "DOI 10.6084/m9.figshare.25018513.v1; file 44102266",
"url": "https://api.figshare.com/v2/articles/25018513",
"published": "2024",
"retrieved_at": "2026-08-27T04:45:09Z",
"media_type": "application/json",
"captured_bytes": 4917,
"captured_sha256": "ce59483bb4dae6871cadf3afa9d650e00d4c524b75083f0c43706436d1220ba6",
"license": "CC BY 4.0 at the article/file level; collection-level license null",
"treatment": "retain file-level scope; do not generalize the license to the collection",
"role": "supplement manifestation; same study root",
"spans": []
},
{
"source_id": "source-illinois-feb-2018",
"work_id": "work-illinois-percentile-charts",
"edition_id": "edition-illinois-feb-2018",
"core": true,
"title": "Illinois February 2018 Bar Examination Percentile Equivalents",
"authors_or_org": "Illinois Board of Admissions to the Bar",
"identifier": "February 2018 official chart",
"url": "https://www.ilbaradmissions.org/percentile-equivalent-charts-feb-2018",
"published": "2018-02",
"retrieved_at": "2026-08-27T05:45:28Z",
"media_type": "application/pdf",
"captured_bytes": 41491,
"captured_sha256": "500d734d54cfaae23b94a988469a8a626fc71abefd5793424bfbe83aabe9e1b7",
"license": "no open license confirmed",
"treatment": "quote only necessary numerical anchors",
"role": "February comparison-population root",
"spans": [
{
"span_id": "span-illinois-feb-2018-anchors",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "total-scale table, rows 300 and 290",
"cells": [
{"cell_id": "cell-illinois-feb-2018-300", "score": 300, "percentile": 90},
{"cell_id": "cell-illinois-feb-2018-290", "score": 290, "percentile": 85}
],
"supports": "The chart brackets 298 but contains no printed 298 row or interpolation rule."
}
]
},
{
"source_id": "source-illinois-jul-2018",
"work_id": "work-illinois-percentile-charts",
"edition_id": "edition-illinois-jul-2018",
"core": true,
"title": "Illinois July 2018 Bar Examination Percentile Equivalents",
"authors_or_org": "Illinois Board of Admissions to the Bar",
"identifier": "July 2018 official chart",
"url": "https://www.ilbaradmissions.org/percentile-equivalent-charts-july-2018",
"published": "2018-07",
"retrieved_at": "2026-08-27T05:45:28Z",
"media_type": "application/pdf",
"captured_bytes": 41047,
"captured_sha256": "9b4251dc1147789eceb9e4e4b3cbdb4e98ab4634f286e8f4d8917d4e0970a299",
"license": "no open license confirmed",
"treatment": "quote only necessary numerical anchors",
"role": "July comparison-population root",
"spans": [
{
"span_id": "span-illinois-jul-2018-anchors",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "total-scale table, rows 300 and 290",
"cells": [
{"cell_id": "cell-illinois-jul-2018-300", "score": 300, "percentile": 70},
{"cell_id": "cell-illinois-jul-2018-290", "score": 290, "percentile": 59}
],
"supports": "The July chart places the same score region far below the February chart."
}
]
},
{
"source_id": "source-illinois-feb-2019",
"work_id": "work-illinois-percentile-charts",
"edition_id": "edition-illinois-feb-2019",
"core": true,
"title": "Illinois February 2019 Bar Examination Percentile Equivalents",
"authors_or_org": "Illinois Board of Admissions to the Bar",
"identifier": "February 2019 official chart",
"url": "https://www.ilbaradmissions.org/percentile-equivalent-charts-february-2019",
"published": "2019-02",
"retrieved_at": "2026-08-27T05:45:29Z",
"media_type": "application/pdf",
"captured_bytes": 40394,
"captured_sha256": "b313a414728db06d9170f79f0177927a3343e3744fc39ba3cadaff1fddc27faa",
"license": "no open license confirmed",
"treatment": "quote only necessary numerical anchors",
"role": "alternate February comparison-population root",
"spans": [
{
"span_id": "span-illinois-feb-2019-anchors",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "total-scale table, rows 300 and 290",
"cells": [
{"cell_id": "cell-illinois-feb-2019-300", "score": 300, "percentile": 90},
{"cell_id": "cell-illinois-feb-2019-290", "score": 290, "percentile": 83}
],
"supports": "This later February chart also brackets 298 but supplies no interpolation rule."
}
]
},
{
"source_id": "source-ncbe-ube-mechanics",
"work_id": "work-ncbe-ube",
"edition_id": "edition-ncbe-ube-scores-2026-08-27",
"core": true,
"title": "UBE Scores",
"authors_or_org": "National Conference of Bar Examiners",
"identifier": "official NCBE legacy UBE scoring page",
"url": "https://www.ncbex.org/exams/ube/ube-scores",
"published": "current page captured 2026-08-27",
"retrieved_at": "2026-08-27T05:45:29Z",
"media_type": "text/html",
"captured_bytes": 63952,
"captured_sha256": "359f8d4e94e128480fe474f07e87a0b019ff20f4cd6041d0b7bfe904b8a84628",
"semantic_capture": {
"normalizer_id": "html-visible-text-root-id-collapse-whitespace-v1",
"command": ["python3", "normalize_html_visible_text.py", "--collapse-whitespace", "--root-id", "block-ncbe-content"],
"root_id": "block-ncbe-content",
"bytes": 1123,
"sha256": "c7484bfc7a063ddde35b93e488f95c0e33894106e5f90df2b66cab9cb18837b7"
},
"license": "no open license confirmed",
"treatment": "quote only necessary scoring mechanics",
"role": "authoritative score-construction root",
"spans": [
{
"span_id": "span-ncbe-ube-weights",
"locator": "legacy UBE scoring overview",
"quote": "The MBE is weighted 50%, the MEE 30%, and the MPT 20%. Legacy UBE total scores are reported on a 400-point scale.",
"supports": "The score is a 400-point weighted composite, not a direct percentile."
}
]
},
{
"source_id": "source-ncbe-mbe-2022",
"work_id": "work-ncbe-2022-statistics",
"edition_id": "edition-ncbe-mbe-2022",
"core": true,
"title": "The Multistate Bar Examination (MBE): 2022 statistics",
"authors_or_org": "National Conference of Bar Examiners",
"identifier": "official 2022 MBE statistics",
"url": "https://thebarexaminer.ncbex.org/2022-statistics/the-multistate-bar-examination-mbe/",
"published": "2023",
"retrieved_at": "2026-08-27T05:45:33Z",
"media_type": "text/html",
"captured_bytes": 278417,
"captured_sha256": "f21c45a6e9d1c1b3a5a538ff4bc5b28751928fac53cb5d174e6ec64488c6b784",
"capture_observations": [
{"role": "author-snapshot", "bytes": 278417, "sha256": "f21c45a6e9d1c1b3a5a538ff4bc5b28751928fac53cb5d174e6ec64488c6b784"},
{"role": "independent-review-snapshot", "bytes": 299423, "sha256": "5cccb74667eb9aed6afc6e958c9ebe82cc5b6f896e15f0463f1758da346e24e6"},
{"role": "author-recapture-2026-08-27", "bytes": 299419, "sha256": "fac79782ff4ca575e75cc1d0cc2bc1f3346f00afaa95570a073f3cad70614541"}
],
"semantic_capture": {
"normalizer_id": "html-visible-text-root-id-collapse-whitespace-v1",
"command": ["python3", "normalize_html_visible_text.py", "--collapse-whitespace", "--root-id", "post-24258"],
"root_id": "post-24258",
"bytes": 4767,
"sha256": "d738481ea3a38e6da39748d47aacc1291112766ec4720917ad0529a02a1f491e"
},
"license": "no open license confirmed",
"treatment": "quote only necessary aggregate statistics",
"role": "authoritative 2022 MBE distribution root",
"spans": [
{
"span_id": "span-ncbe-mbe-2022-counts",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "2022 MBE National Summary Statistics",
"cells": [
{"cell_id": "cell-ncbe-mbe-count-february", "row": "Number of Examinees", "column": "February", "text": "16,504"},
{"cell_id": "cell-ncbe-mbe-count-july", "row": "Number of Examinees", "column": "July", "text": "44,705"},
{"cell_id": "cell-ncbe-mbe-count-overall", "row": "Number of Examinees", "column": "2022 Overall", "text": "61,209"},
{"cell_id": "cell-ncbe-mbe-mean-february", "row": "Mean Scaled Score", "column": "February", "text": "132.6"},
{"cell_id": "cell-ncbe-mbe-mean-july", "row": "Mean Scaled Score", "column": "July", "text": "140.3"},
{"cell_id": "cell-ncbe-mbe-mean-overall", "row": "Mean Scaled Score", "column": "2022 Overall", "text": "138.3"},
{"cell_id": "cell-ncbe-mbe-sd-february", "row": "Standard Deviation", "column": "February", "text": "15.4"},
{"cell_id": "cell-ncbe-mbe-sd-july", "row": "Standard Deviation", "column": "July", "text": "17.0"},
{"cell_id": "cell-ncbe-mbe-sd-overall", "row": "Standard Deviation", "column": "2022 Overall", "text": "17.0"}
],
"supports": "February, July, and overall MBE populations differ in size and distribution."
}
]
},
{
"source_id": "source-ncbe-snapshot-2022",
"work_id": "work-ncbe-2022-statistics",
"edition_id": "edition-ncbe-snapshot-2022",
"core": true,
"title": "2022 Statistics Snapshot",
"authors_or_org": "National Conference of Bar Examiners",
"identifier": "official 2022 statistics snapshot",
"url": "https://thebarexaminer.ncbex.org/2022-statistics/2022-statistics-snapshot/",
"published": "2023",
"retrieved_at": "2026-08-27T05:45:33Z",
"media_type": "text/html",
"captured_bytes": 188472,
"captured_sha256": "391deeac882bfe4dfb69a58e852c465ffbcc9e1d2e0a8557208f82cf795309a0",
"capture_observations": [
{"role": "author-snapshot", "bytes": 188472, "sha256": "391deeac882bfe4dfb69a58e852c465ffbcc9e1d2e0a8557208f82cf795309a0"},
{"role": "independent-review-snapshot", "bytes": 188472, "sha256": "6417676e0a1efb43fdf7a0014e45f253b14e63fd14097d1bc9b2e77699c8c728"},
{"role": "author-recapture-2026-08-27", "bytes": 188472, "sha256": "1343b00bd6f500d7309b137b8dbbe98765a824e32b280a21dc3201f7f7110c2d"}
],
"semantic_capture": {
"normalizer_id": "html-visible-text-root-id-collapse-whitespace-v1",
"command": ["python3", "normalize_html_visible_text.py", "--collapse-whitespace", "--root-id", "post-24228"],
"root_id": "post-24228",
"bytes": 5449,
"sha256": "d1373f31770be2bbecb821047ef0ea95cb0c50e6088e0c26322de921cf797917"
},
"license": "no open license confirmed",
"treatment": "quote only necessary population aggregates",
"role": "first-time/repeater composition root",
"spans": [
{
"span_id": "span-ncbe-snapshot-composition",
"format": "exact-segments",
"normalization": "collapse-whitespace",
"locator": "NCBE MBE-based data, February and July 2022 totals",
"segments": [
{"segment_id": "segment-ncbe-february-repeaters-count", "text": "February likely repeaters taking: 11,289"},
{"segment_id": "segment-ncbe-february-repeaters-percent", "text": "68% of February 2022 examinees were likely repeaters"},
{"segment_id": "segment-ncbe-february-first-timers-count", "text": "February likely first-timers taking: 5,215"},
{"segment_id": "segment-ncbe-february-first-timers-percent", "text": "32% of February 2022 examinees were likely first-time takers"},
{"segment_id": "segment-ncbe-july-repeaters-count", "text": "July likely repeaters taking: 10,200"},
{"segment_id": "segment-ncbe-july-repeaters-percent", "text": "23% of July 2022 examinees were likely repeaters"},
{"segment_id": "segment-ncbe-july-first-timers-count", "text": "July likely first-timers taking: 34,505"},
{"segment_id": "segment-ncbe-july-first-timers-percent", "text": "77% of July 2021 examinees were likely first-time takers"}
],
"supports": "February and July have materially different inferred first-time/repeater composition."
},
{
"span_id": "span-ncbe-snapshot-typo",
"locator": "NCBE MBE-based data, July 2022 section",
"quote": "77% of July 2021 examinees were likely first-time takers",
"supports": "The page's July 2022 section contains an apparent 2021 year-label typo that must not be silently corrected in quotation."
}
]
},
{
"source_id": "source-ncbe-first-repeat-2022",
"work_id": "work-ncbe-2022-statistics",
"edition_id": "edition-ncbe-first-repeat-2022",
"core": true,
"title": "First-Time Exam Takers and Repeaters in 2022",
"authors_or_org": "National Conference of Bar Examiners",
"identifier": "official 2022 jurisdiction-reported table",
"url": "https://thebarexaminer.ncbex.org/2022-statistics/first-time-exam-takers-and-repeaters-in-2022/",
"published": "2023",
"retrieved_at": "2026-08-27T05:45:35Z",
"media_type": "text/html",
"captured_bytes": 372700,
"captured_sha256": "d5edc89f2c781ab7a795602cdc4a6b42b3da457f13972c8660db9a2b976df448",
"capture_observations": [
{"role": "author-snapshot", "bytes": 372700, "sha256": "d5edc89f2c781ab7a795602cdc4a6b42b3da457f13972c8660db9a2b976df448"},
{"role": "independent-review-snapshot", "bytes": 372700, "sha256": "6a2701bcd45855deaefcb1c7e4437125765adc756c4f193280c1f338691ad69c"},
{"role": "author-recapture-2026-08-27", "bytes": 372700, "sha256": "0c6f9b9ac682e46ccd6961425dbb613d05833268733d76fbea2b189d4750bd20"}
],
"semantic_capture": {
"normalizer_id": "html-visible-text-root-id-collapse-whitespace-v1",
"command": ["python3", "normalize_html_visible_text.py", "--collapse-whitespace", "--root-id", "post-24476"],
"root_id": "post-24476",
"bytes": 18887,
"sha256": "28a6ddf863f104d59b23674ed11a46951b87e088f4774c277f642e6d077e582d"
},
"license": "no open license confirmed",
"treatment": "quote only necessary methodological boundary",
"role": "jurisdiction-reported status boundary",
"spans": [
{
"span_id": "span-ncbe-jurisdiction-status",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "table note",
"quote": "NOTE: First-time exam and repeat test taker data supplied by the jurisdictions in this chart are based on those examinees’ testing experience in the reporting jurisdiction only and do not account for possible previous attempts at the bar examination in other jurisdictions.",
"supports": "Jurisdiction-reported status is not the same denominator as NCBE's cross-jurisdiction MBE-based classification."
}
]
},
{
"source_id": "source-martinez-vor",
"work_id": "work-martinez-reanalysis",
"edition_id": "edition-martinez-2024-vor",
"core": true,
"title": "Re-evaluating GPT-4's bar exam performance",
"authors_or_org": "Eric Martínez",
"identifier": "DOI 10.1007/s10506-024-09396-9",
"url": "https://scholarship.law.tamu.edu/cgi/viewcontent.cgi?article=3387&context=facscholar",
"landing_url": "https://scholarship.law.tamu.edu/facscholar/2405/",
"publisher_url": "https://link.springer.com/content/pdf/10.1007/s10506-024-09396-9.pdf",
"published": "2024-03-30",
"retrieved_at": "2026-08-23T17:24:01Z",
"media_type": "application/pdf",
"captured_bytes": 888663,
"captured_sha256": "bbab759cb88e93a5216936af1edb2726eb8eb0edde3148d18c1093bed9226e76",
"carrier": {
"carrier_id": "carrier-tamu-facscholar-3387",
"institution": "Texas A&M University School of Law",
"page_count": 25,
"journal_page_range": "581-604",
"landing_capture": {
"retrieved_at": "2026-08-23T17:23:50Z",
"bytes": 40754,
"sha256": "d28416c1a0681b281062d47c811e14013b822399eb0fb2abdf323d0c9841517f"
},
"span_readback_ids": [
"span-martinez-abstract-ranks",
"span-martinez-table-45",
"span-martinez-model-assumptions",
"span-martinez-mean-assumption",
"span-martinez-results-45",
"span-martinez-discussion-48",
"span-martinez-score-validation"
],
"retrieval_limitation": "The institutional landing page remains public and identifies the PDF, DOI, pages, and CC BY 4.0 rights; automated direct-PDF refreshes can return HTTP 403, so the exact credential-free 2026-08-23 capture is retained by digest."
},
"license": "CC BY 4.0",
"treatment": "attributed quotation permitted; retain assumptions and internal conflicts",
"role": "distinct analytical root reusing the reported score",
"spans": [
{
"span_id": "span-martinez-abstract-ranks",
"format": "exact-segments",
"normalization": "collapse-whitespace",
"locator": "abstract, journal p. 581",
"segments": [
{
"segment_id": "segment-martinez-abstract-first-time",
"text": "Third, examining official NCBE data and using several conservative statistical assumptions, GPT-4’s performance against first-time test takers is estimated to be ∼62nd percentile, including ∼42nd percentile on essays."
},
{
"segment_id": "segment-martinez-abstract-passers",
"text": "Fourth, when examining only those who passed the exam (i.e. licensed or license-pending attorneys), GPT-4’s performance is estimated to drop to ∼48th percentile overall, and ∼15th percentile on essays."
}
],
"supports": "The abstract reports modeled 62nd and 48th ranks for distinct comparison populations."
},
{
"span_id": "span-martinez-table-45",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "Table 3, journal p. 591",
"cells": [
{"cell_id": "cell-martinez-july-ube", "row": "July test-takers", "column": "UBE", "text": "1st–68th"},
{"cell_id": "cell-martinez-first-timers-ube", "row": "All first-timers", "column": "UBE", "text": "2nd–62rd"},
{"cell_id": "cell-martinez-qualified-attorneys-ube", "row": "Qualified attorneys", "column": "UBE", "text": "0th–45th"}
],
"supports": "The table reports 45th, not 48th, for the passers/qualified-attorneys comparison."
},
{
"span_id": "span-martinez-model-assumptions",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "§3.1, journal pp. 588-589",
"quote": "Assuming that UBE scores (as well as MBE and essay subscores) are normally distributed, percentiles of GPT’s score can be directly computed after computing the parameters of these distributions (i.e. the mean and standard deviation).",
"supports": "The re-analysis is model-based and depends on a normality assumption."
},
{
"span_id": "span-martinez-mean-assumption",
"format": "exact-segments",
"normalization": "collapse-whitespace",
"locator": "§3.1, journal p. 588",
"segments": [
{"segment_id": "segment-martinez-essay-mean", "text": "Thus, the methodology here assumed that the mean first-time essay score is 143.8."},
{"segment_id": "segment-martinez-ube-mean", "text": "Given that the total UBE score is computed directly by adding MBE and essay scores (National Conference of Bar Examiners n.d.-h), an assumption was made that mean first-time UBE score is 287.6 (143.8 + 143.8)."}
],
"supports": "The UBE mean is derived from an assumed essay mean, not directly observed as a national UBE mean."
},
{
"span_id": "span-martinez-results-45",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "§3.2.2, journal p. 591",
"quote": "With regard to the aggregate UBE score, GPT-4 scored in the ∼45th percentile.",
"supports": "The results section reports 45th among those who passed."
},
{
"span_id": "span-martinez-discussion-48",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "discussion, journal p. 598",
"quote": "when examining only those who passed the exam, GPT-4’s performance is estimated to drop to ∼48th percentile overall, and ∼15th percentile on essays.",
"supports": "The discussion reports 48th, creating an internal 45th-versus-48th conflict."
},
{
"span_id": "span-martinez-score-validation",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "introduction, journal p. 584",
"quote": "The paper successfully replicates the MBE score of 158, but highlights several methodological issues in the grading of the MPT + MEE components of the exam, which call into question the validity of the essay score (140).",
"supports": "The MBE and author-graded essay components have different validation status."
}
]
},
{
"source_id": "source-martinez-osf",
"work_id": "work-martinez-reanalysis",
"edition_id": "edition-martinez-osf-c8ygu",
"core": true,
"title": "Re-evaluating GPT-4's bar exam performance analysis deposit",
"authors_or_org": "Eric Martínez",
"identifier": "OSF node c8ygu; anonymous view token dcc617accc464491922b77414867a066",
"url": "https://api.osf.io/v2/nodes/c8ygu/?view_only=dcc617accc464491922b77414867a066",
"published": "2024",
"retrieved_at": "2026-08-27T04:45:11Z",
"media_type": "application/json",
"captured_bytes": 3696,
"captured_sha256": "06c756ef9d72bce0e92e821917bd5ca02fa1d5089b965689a2dd4f4d28f15bf5",
"capture_observations": [
{
"role": "author-snapshot",
"bytes": 3696,
"sha256": "06c756ef9d72bce0e92e821917bd5ca02fa1d5089b965689a2dd4f4d28f15bf5"
},
{
"role": "independent-snapshot",
"bytes": 3697,
"sha256": "b18212bf119cd65f826d137a1306a266b446de193618822fcdcea3eae0903a6f"
}
],
"semantic_capture": {
"normalizer_id": "canonical-json-v1",
"command": ["python", "-m", "json.tool", "--sort-keys"],
"bytes": 3697,
"sha256": "b18212bf119cd65f826d137a1306a266b446de193618822fcdcea3eae0903a6f"
},
"license": "no node or file license confirmed",
"treatment": "metadata and quote-minimal script spans only; no redistribution",
"role": "analysis-code manifestation; same re-analysis root",
"spans": []
},
{
"source_id": "source-ncbe-ube-2022",
"work_id": "work-ncbe-2022-statistics",
"edition_id": "edition-ncbe-ube-2022",
"core": false,
"title": "The Uniform Bar Examination (UBE): 2022 statistics",
"authors_or_org": "National Conference of Bar Examiners",
"identifier": "official 2022 UBE statistics",
"url": "https://thebarexaminer.ncbex.org/2022-statistics/the-uniform-bar-examination-ube/",
"published": "2023",
"retrieved_at": "2026-08-27T14:57:28Z",
"media_type": "text/html",
"captured_bytes": 260280,
"captured_sha256": "eb3e0b45cd4496cfc15669c267f25107c27e31489ece46dd43def1356a966e52",
"license": "no open license confirmed",
"treatment": "quote only the historical jurisdiction cutoff",
"role": "supplemental cutoff source",
"spans": [
{
"span_id": "span-ncbe-ny-cutoff-2022",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "Minimum Passing UBE Score by Jurisdiction in 2022, row 266",
"cells": [
{"cell_id": "cell-ncbe-ube-score-266", "row": "266", "column": "Score", "text": "266"},
{"cell_id": "cell-ncbe-ube-jurisdictions-266", "row": "266", "column": "Jurisdictions", "text": "Connecticut; District of Columbia; Illinois; Iowa; Kansas; Kentucky; Maryland; Montana; New Jersey; New York; South Carolina; Virgin Islands"}
],
"supports": "New York's 2022 minimum passing UBE score was 266."
}
]
},
{
"source_id": "source-ny-passrates-2022",
"work_id": "work-ny-2022-passrates",
"edition_id": "edition-ny-passrates-2022",
"core": false,
"title": "New York Bar Exam 2022 Statistics",
"authors_or_org": "New York State Board of Law Examiners",
"identifier": "2022_NY_Bar_Exam_PassRates.pdf",
"url": "https://www.nybarexam.org/ExamStats/2022_NY_Bar_Exam_PassRates.pdf",
"published": "2022",
"retrieved_at": "2026-08-27T14:40:54Z",
"media_type": "application/pdf",
"captured_bytes": 65751,
"captured_sha256": "9cddfa2dbe71b11e01a5ee24a328952cdb8c6a81ee140e83a8a1713aa4c17088",
"license": "no open license confirmed",
"treatment": "quote only necessary aggregate counts",
"role": "supplemental pass-rate input",
"spans": [
{
"span_id": "span-ny-first-timers-2022",
"format": "table-cell-transcription",
"normalization": "exact-cell-text",
"locator": "FIRST TIMERS, ALL Candidates, Combined 2022",
"cells": [
{"cell_id": "cell-ny-first-timers-took", "row": "ALL Candidates", "column": "Took", "text": "9,457"},
{"cell_id": "cell-ny-first-timers-passed", "row": "ALL Candidates", "column": "Passed", "text": "6,867"},
{"cell_id": "cell-ny-first-timers-rate", "row": "ALL Candidates", "column": "Rate", "text": "73%"}
],
"supports": "The combined first-time pass rate was 73%, so the complementary non-pass proportion used by the re-analysis is 27%."
}
]
},
{
"source_id": "source-reshetar-testing-column-2022",
"work_id": "work-reshetar-testing-column-february-july",
"edition_id": "edition-reshetar-testing-column-spring-2022",
"core": false,
"title": "The Testing Column: Why Are February Bar Exam Pass Rates Lower than July Pass Rates?",
"authors_or_org": "Rosemary Reshetar, EdD / National Conference of Bar Examiners",
"attribution_conflict": {
"visible_print_byline": "Rosemary Reshetar, EdD",
"html_json_ld_author": "Jim Leach",
"treatment": "Use the visible print byline for work attribution; retain the conflicting JSON-LD site metadata without silently treating it as authorship."
},
"identifier": "The Bar Examiner, Spring 2022 (Vol. 91, No. 1), pp. 51-53",
"url": "https://thebarexaminer.ncbex.org/article/spring-2022/the-testing-column-5/",
"published": "2022-03",
"retrieved_at": "2026-08-27T14:40:54Z",
"media_type": "text/html",
"captured_bytes": 218131,
"captured_sha256": "174a5adc50f0235f70ebea3b6c4ed85fbe91df8c18d8144286876fe6b95e70bd",
"capture_observations": [
{
"role": "author-snapshot",
"bytes": 218131,
"sha256": "174a5adc50f0235f70ebea3b6c4ed85fbe91df8c18d8144286876fe6b95e70bd"
},
{
"role": "independent-snapshot",
"bytes": 218131,
"sha256": "ca8dd034a3a41f43fd177ce493cf1489f894c3ae65610033d4e3c26a472a83f3"
},
{
"role": "changes-required-reviewer",
"bytes": 218131,
"sha256": "3e0af4b53f984113fd347ba426d665ad5b3cfbedb3505013fae55d219877423a"
},
{
"role": "correction-author",
"bytes": 218131,
"sha256": "0707ff4563ca7969d8ee11c136da9c15e195ee7d827199b58ed95d6a10f4d7ab"
}
],
"semantic_capture": {
"normalizer_id": "html-visible-text-root-id-collapse-whitespace-v1",
"command": ["python3", "normalize_html_visible_text.py", "--collapse-whitespace", "--root-id", "post-22614"],
"root_id": "post-22614",
"bytes": 12498,
"sha256": "2222ea3b0110bcc02c189bb5d79ee9264ada557b2edae09077d985d472da92f6"
},
"license": "no open license confirmed",
"treatment": "quote only necessary group means",
"role": "supplemental first-time MBE mean input",
"spans": [
{
"span_id": "span-reshetar-first-time-mean",
"format": "exact-contiguous-text",
"normalization": "collapse-whitespace",
"locator": "paragraph beginning 'The numbers from 2019'",
"quote": "The numbers from 2019, the last prepandemic year, are typical: all likely first-time test takers earned an average MBE score of 143.8. Likely repeaters, however, earned an average MBE score of 132.4.",
"supports": "The re-analysis uses 143.8 as the first-time MBE mean."
}
]
},
{
"source_id": "source-martinez-analysis-new",
"work_id": "work-martinez-reanalysis",
"edition_id": "edition-martinez-osf-analysis-new",
"core": false,
"title": "percentile_analysis_new.Rmd",
"authors_or_org": "Eric Martínez",
"identifier": "OSF file gujmp; SHA-256 e73b60d3bcba8075b8e513f53d079541ca11b32fa68a8a1201a6442644add588",
"url": "https://osf.io/download/gujmp/?view_only=dcc617accc464491922b77414867a066",
"published": "2024",
"retrieved_at": "2026-08-27T05:48:35Z",
"media_type": "text/plain",
"captured_bytes": 12730,
"captured_sha256": "e73b60d3bcba8075b8e513f53d079541ca11b32fa68a8a1201a6442644add588",
"license": "no file license confirmed",
"treatment": "quote-minimal code readback only; do not redistribute the full file",
"role": "supplemental derivation implementation",
"spans": [
{
"span_id": "span-martinez-script-july-mbe-distribution",
"format": "code-table-transcription",
"normalization": "exact-code-text",
"locator": "lines 55-58",
"code_lines": [
{"line_id": "line-martinez-july-distribution-55", "text": "July_distribution <- c(rep(85, 2), rep(90, 2), rep(95, 5), rep(100, 6), rep(105, 13),"},
{"line_id": "line-martinez-july-distribution-56", "text": " rep(110, 22), rep(115, 33), rep(120, 56), rep(125, 73), rep(130, 78),"},
{"line_id": "line-martinez-july-distribution-57", "text": " rep(135, 104), rep(140, 96), rep(145, 101), rep(150, 99), rep(155, 99),"},
{"line_id": "line-martinez-july-distribution-58", "text": " rep(160, 79), rep(165, 64), rep(170, 38), rep(175, 22), rep(180, 8), rep(185, 2))"}
],
"cells": [
{"cell_id": "cell-martinez-july-mbe-85", "score": 85, "count": 2},
{"cell_id": "cell-martinez-july-mbe-90", "score": 90, "count": 2},
{"cell_id": "cell-martinez-july-mbe-95", "score": 95, "count": 5},
{"cell_id": "cell-martinez-july-mbe-100", "score": 100, "count": 6},
{"cell_id": "cell-martinez-july-mbe-105", "score": 105, "count": 13},
{"cell_id": "cell-martinez-july-mbe-110", "score": 110, "count": 22},
{"cell_id": "cell-martinez-july-mbe-115", "score": 115, "count": 33},
{"cell_id": "cell-martinez-july-mbe-120", "score": 120, "count": 56},
{"cell_id": "cell-martinez-july-mbe-125", "score": 125, "count": 73},
{"cell_id": "cell-martinez-july-mbe-130", "score": 130, "count": 78},
{"cell_id": "cell-martinez-july-mbe-135", "score": 135, "count": 104},
{"cell_id": "cell-martinez-july-mbe-140", "score": 140, "count": 96},
{"cell_id": "cell-martinez-july-mbe-145", "score": 145, "count": 101},
{"cell_id": "cell-martinez-july-mbe-150", "score": 150, "count": 99},
{"cell_id": "cell-martinez-july-mbe-155", "score": 155, "count": 99},
{"cell_id": "cell-martinez-july-mbe-160", "score": 160, "count": 79},
{"cell_id": "cell-martinez-july-mbe-165", "score": 165, "count": 64},
{"cell_id": "cell-martinez-july-mbe-170", "score": 170, "count": 38},
{"cell_id": "cell-martinez-july-mbe-175", "score": 175, "count": 22},
{"cell_id": "cell-martinez-july-mbe-180", "score": 180, "count": 8},
{"cell_id": "cell-martinez-july-mbe-185", "score": 185, "count": 2}
],
"supports": "The executable analysis expands these 21 score-count cells to estimate the July MBE sample standard deviation."
},
{
"span_id": "span-martinez-script-ube-sd",
"format": "code-segment-transcription",
"normalization": "exact-code-text",
"locator": "lines 104-112",
"segments": [
{"segment_id": "segment-martinez-code-mean", "text": "mean <- 287.6"},
{"segment_id": "segment-martinez-code-quantile", "text": "quantile_value <- 266"},
{"segment_id": "segment-martinez-code-percentile", "text": "percentile <- 0.27 # 24%"},
{"segment_id": "segment-martinez-code-z-score", "text": "z_score <- qnorm(percentile)"},
{"segment_id": "segment-martinez-code-ube-sd", "text": "sd_UBE <- (quantile_value - mean) / z_score"}
],
"supports": "The script derives UBE standard deviation from mean 287.6, cutoff 266, and non-pass proportion 0.27."
},
{
"span_id": "span-martinez-script-comment-conflict",
"format": "code-segment-transcription",
"normalization": "exact-code-text",
"locator": "lines 106-109",
"segments": [
{"segment_id": "segment-martinez-code-comment-value", "text": "percentile <- 0.27 # 24%"},
{"segment_id": "segment-martinez-code-comment-prose", "text": "# Calculate the z-score for the 24th percentile"}
],
"supports": "The executable value is 0.27 while adjacent comments say 24%, an internal code-comment discrepancy."
},
{
"span_id": "span-martinez-script-thresholds",
"format": "code-segment-transcription",
"normalization": "exact-code-text",
"locator": "lines 166-168",
"segments": [
{"segment_id": "segment-martinez-code-filter-ube", "text": "filtered_data_ube <- data_ube[data_ube >= 270]"},
{"segment_id": "segment-martinez-code-filter-mbe", "text": "filtered_data_mbe <- data_mbe[data_mbe >= 135]"},
{"segment_id": "segment-martinez-code-filter-essay", "text": "filtered_data_essay <- data_essay[data_essay >= 135]"}
],
"supports": "The passers analysis filters UBE at 270 even though the SD input uses New York's 266 cutoff."
}
]
}
],
"claims": [
{
"claim_id": "claim-launch-score-label",
"proposition": "OpenAI's launch-edition report displayed 298/400 and approximately 90th percentile for a simulated UBE.",
"kind": "reported-assertion",
"span_ids": ["span-openai-v1-table-score"]
},
{
"claim_id": "claim-launch-comparison-unspecified",
"proposition": "The launch report says test takers but does not disclose a UBE administration, jurisdiction, population composition, chart, or interpolation for the displayed percentile.",
"kind": "bounded-negative-result",
"span_ids": ["span-openai-v1-scoring", "span-katz-vor-percentile-boundary"],
"negative_search_id": "search-launch-percentile-provenance"
},
{
"claim_id": "claim-score-discrepancy",
"proposition": "The report's 298 and study's approximately 297 are distinct scoring choices within one underlying experiment, not independent performance roots.",
"kind": "lineage-qualified-synthesis",
"span_ids": ["span-openai-v1-table-score", "span-katz-vor-abstract-score", "span-katz-vor-score-discrepancy"]
},
{
"claim_id": "claim-february-sensitive",
"proposition": "A linear interpolation not supplied by the sources places 298 near 89th in the February 2018 chart and 88.6th in the February 2019 chart.",
"kind": "reviewer-derived-sensitivity",
"span_ids": ["span-illinois-feb-2018-anchors", "span-illinois-feb-2019-anchors"],
"derivation_ids": ["derive-illinois-feb-2018-298", "derive-illinois-feb-2019-298"]
},
{
"claim_id": "claim-july-sensitive",
"proposition": "The same disclosed linear interpolation places 298 near 67.8th in the July 2018 Illinois chart.",
"kind": "reviewer-derived-sensitivity",
"span_ids": ["span-illinois-jul-2018-anchors"],
"derivation_ids": ["derive-illinois-jul-2018-298"]
},
{
"claim_id": "claim-martinez-first-time",
"proposition": "Under Martínez's modeled assumptions, 298 is approximately 62nd percentile among first-time takers.",
"kind": "assumption-bound-derived-result",
"span_ids": ["span-martinez-model-assumptions", "span-martinez-mean-assumption", "span-ny-first-timers-2022", "span-ncbe-ny-cutoff-2022", "span-martinez-script-ube-sd"],
"derivation_ids": ["derive-martinez-first-time-ube"]
},
{
"claim_id": "claim-martinez-passers-conflict",
"proposition": "The Martínez version of record and code support roughly 45th under the encoded passers model, while its abstract and discussion say roughly 48th; the packet preserves 45/48 as an internal discrepancy.",
"kind": "within-edition-conflict",
"span_ids": ["span-martinez-table-45", "span-martinez-results-45", "span-martinez-discussion-48", "span-martinez-script-thresholds"],
"derivation_ids": ["derive-martinez-passers-ube"]
},
{
"claim_id": "claim-no-lawyer-rank",
"proposition": "None of the captured sources measures performance against practicing lawyers or establishes general legal competence.",
"kind": "scope-boundary",
"span_ids": ["span-openai-v1-abstract-top-ten", "span-martinez-model-assumptions"]
}
],
"lineages": [
{
"lineage_id": "lineage-model-performance-root",
"root_type": "performance",
"unit": "one historical simulated-UBE experiment assembled from multiple-choice and author-graded free-response components",
"source_ids": ["source-openai-v1", "source-openai-v6", "source-katz-vor", "source-katz-ssrn", "source-katz-git-snapshot", "source-katz-figshare"],
"independent_roots": 1,
"dependence": ["shared model snapshots", "shared prompts and purchased/public exam material", "shared score outputs", "OpenAI collaboration with study authors", "report/preprint/VOR/repository/supplement manifestations"]
},
{
"lineage_id": "lineage-martinez-analysis-root",
"root_type": "analysis",
"unit": "one re-analysis of the reported historical score against alternate aggregate comparison populations",
"source_ids": ["source-martinez-vor", "source-martinez-osf", "source-martinez-analysis-new"],
"independent_roots": 1,
"dependence": ["reuses the reported OpenAI/Katz score", "reuses Illinois and NCBE aggregates", "VOR and OSF are manifestations of the same analysis"]
},
{
"lineage_id": "lineage-illinois-comparison-root",
"root_type": "comparison-data",
"unit": "official administration-specific Illinois aggregate charts",
"source_ids": ["source-illinois-feb-2018", "source-illinois-jul-2018", "source-illinois-feb-2019"],
"independent_roots": 3,
"dependence": ["same state scoring system", "different administrations and population composition", "no disclosed 297/298 row or interpolation"]
},
{
"lineage_id": "lineage-ncbe-comparison-root",
"root_type": "comparison-data",
"unit": "official NCBE scoring and population aggregates",
"source_ids": ["source-ncbe-ube-mechanics", "source-ncbe-mbe-2022", "source-ncbe-snapshot-2022", "source-ncbe-first-repeat-2022", "source-ncbe-ube-2022", "source-reshetar-testing-column-2022"],
"independent_roots": 1,
"dependence": ["multiple tables and explanatory pages from one statistical authority", "jurisdiction-reported and NCBE MBE-based first-time classifications differ"]
},
{
"lineage_id": "lineage-new-york-pass-rate-root",
"root_type": "comparison-data",
"unit": "New York 2022 first-time pass counts",
"source_ids": ["source-ny-passrates-2022"],
"independent_roots": 1,
"dependence": ["same jurisdiction and year used to parameterize the Martínez model"]
}
],
"lineage_edges": [
{
"edge_id": "edge-data-reported-score-reuse",
"edge_type": "data",
"from_ids": ["lineage-model-performance-root"],
"to_ids": ["lineage-martinez-analysis-root"],
"evidence": [
{
"source_id": "source-katz-vor",
"span_ids": ["span-katz-vor-components"],
"finding": "The re-analysis inherits the reported historical component and composite scores.",
"independence_effect": "The re-analysis is not a new model-performance experiment."
},
{
"source_id": "source-martinez-vor",
"span_ids": ["span-martinez-score-validation"],
"finding": "Martínez validates the inherited MBE score while questioning the inherited essay score.",
"independence_effect": "Score validation changes warrant, not the participant or model-performance root count."
}
]
},
{
"edge_id": "edge-model-historical-snapshots",
"edge_type": "model",
"from_ids": ["source-openai-v1"],
"to_ids": ["lineage-model-performance-root"],
"evidence": [
{
"source_id": "source-openai-v1",
"span_ids": ["span-openai-v1-snapshots"],
"finding": "The composite used two dated GPT-4 snapshots.",
"independence_effect": "The two snapshots are components of one reported experiment, not two evidence roots."
}
]
},
{
"edge_id": "edge-author-social-collaboration",
"edge_type": "author-social",
"from_ids": ["source-openai-v1"],
"to_ids": ["lineage-model-performance-root"],
"evidence": [
{
"source_id": "source-openai-v1",
"span_ids": ["span-openai-v1-collaborators"],
"finding": "OpenAI identifies CaseText and Stanford CodeX as collaborators who ran the UBE.",
"independence_effect": "The vendor report and study are socially linked manifestations, not independent tests."
}
]
},
{
"edge_id": "edge-method-single-free-response-run",
"edge_type": "method",
"from_ids": ["source-openai-v1"],
"to_ids": ["lineage-model-performance-root"],
"evidence": [
{
"source_id": "source-openai-v1",
"span_ids": ["span-openai-v1-free-response-run"],
"finding": "Each free-response question was run once under one best-guess setting.",
"independence_effect": "Repeated documents do not supply repeated-run independence."
}
]
},
{
"edge_id": "edge-material-shared-exam-items",
"edge_type": "material",
"from_ids": ["source-katz-vor"],
"to_ids": ["lineage-model-performance-root"],
"evidence": [
{
"source_id": "source-katz-vor",
"span_ids": ["span-katz-vor-materials"],
"finding": "The study used July 2022 MEE and MPT questions and official MBE questions from prior administrations.",
"independence_effect": "All score manifestations inherit the same exam-item materials."
}
]
},
{
"edge_id": "edge-benchmark-illinois-charts",
"edge_type": "benchmark",
"from_ids": ["lineage-illinois-comparison-root"],
"to_ids": ["lineage-martinez-analysis-root"],
"evidence": [
{
"source_id": "source-illinois-feb-2018",
"span_ids": ["span-illinois-feb-2018-anchors"],
"finding": "The February 2018 chart supplies administration-specific anchor cells.",
"independence_effect": "It is comparison data, not an additional GPT-4 performance root."
},
{
"source_id": "source-illinois-jul-2018",
"span_ids": ["span-illinois-jul-2018-anchors"],
"finding": "The July 2018 chart supplies a different administration-specific comparison.",
"independence_effect": "Population drift affects rank without changing the model score."
},
{
"source_id": "source-illinois-feb-2019",
"span_ids": ["span-illinois-feb-2019-anchors"],
"finding": "The February 2019 chart supplies another administration-specific comparison.",
"independence_effect": "It is a sensitivity input, not a replication of the model run."
}
]
},
{
"edge_id": "edge-score-component-composite",
"edge_type": "score",
"from_ids": ["source-katz-vor"],
"to_ids": ["lineage-model-performance-root"],
"evidence": [
{
"source_id": "source-katz-vor",
"span_ids": ["span-katz-vor-components", "span-katz-vor-score-discrepancy"],
"finding": "The VOR binds component scores to 297 and explains how an alternate MBE choice yields 298 or higher.",
"independence_effect": "297 and 298 are scoring variants within one experiment."
}
]
},
{
"edge_id": "edge-comparison-class-first-time-passers",
"edge_type": "comparison-class",
"from_ids": ["lineage-ncbe-comparison-root", "lineage-new-york-pass-rate-root"],
"to_ids": ["lineage-martinez-analysis-root"],
"evidence": [
{
"source_id": "source-ncbe-first-repeat-2022",
"span_ids": ["span-ncbe-jurisdiction-status"],
"finding": "Jurisdiction-reported first-time status is explicitly jurisdiction-bounded.",
"independence_effect": "Comparison populations are not interchangeable."
},
{
"source_id": "source-ncbe-snapshot-2022",
"span_ids": ["span-ncbe-snapshot-composition"],
"finding": "February and July populations have materially different inferred first-time and repeater composition.",
"independence_effect": "Administration mix can change percentile without new model evidence."
},
{
"source_id": "source-ny-passrates-2022",
"span_ids": ["span-ny-first-timers-2022"],
"finding": "The New York first-time table supplies the pass-rate complement used by the model.",
"independence_effect": "It parameterizes the comparison model rather than testing GPT-4."
},
{
"source_id": "source-ncbe-ube-2022",
"span_ids": ["span-ncbe-ny-cutoff-2022"],
"finding": "The official table places New York in the 266 minimum-score row.",
"independence_effect": "The threshold defines a modeled comparator boundary."
}
]
},
{
"edge_id": "edge-citation-katz-to-martinez",
"edge_type": "citation",
"from_ids": ["lineage-model-performance-root"],
"to_ids": ["lineage-martinez-analysis-root"],
"evidence": [
{
"source_id": "source-katz-vor",
"span_ids": ["span-katz-vor-range"],
"finding": "The VOR expressly discusses later analysis while bounding the raw score to a population-sensitive range.",
"independence_effect": "Citation links the analysis lineage to the same reported score."
}
]
},
{
"edge_id": "edge-derivation-comparison-inputs",
"edge_type": "derivation",
"from_ids": ["lineage-ncbe-comparison-root", "lineage-new-york-pass-rate-root"],
"to_ids": ["lineage-martinez-analysis-root"],
"evidence": [
{
"source_id": "source-reshetar-testing-column-2022",
"span_ids": ["span-reshetar-first-time-mean"],
"finding": "The Reshetar column supplies the 143.8 first-time MBE mean; conflicting JSON-LD names Jim Leach only as site metadata.",
"independence_effect": "The value is a modeled input, not a GPT-4 observation."
},
{
"source_id": "source-martinez-analysis-new",
"span_ids": ["span-martinez-script-july-mbe-distribution", "span-martinez-script-ube-sd", "span-martinez-script-thresholds"],
"finding": "The script binds the July MBE bins, UBE SD construction, and passers thresholds.",
"independence_effect": "Executable transformations do not create new source or performance roots."
}
]
}
],
"negative_searches": [
{
"negative_search_id": "search-launch-percentile-provenance",
"git_blob_search_manifest_id": "em:git-blob-search:sha256:545908f30ba849c42c860185f92612f4d52a53b4f13b3c3ca5672213e23ba996",
"target": "the exact chart, administration, denominator, and interpolation used for OpenAI's launch-edition approximately-90th UBE label",
"bounded_scope": [
"OpenAI arXiv v1 report, references, and exam-method appendices",
"OpenAI arXiv v6 drift check",
"Katz SSRN/VOR and footnotes",
"all 78 blobs in the pinned Katz Git tree: every body SHA-256-bound, all 72 UTF-8 bodies searched, and 6 binary bodies retained as digested no-text-search records",
"the one Figshare supplement",
"Martínez VOR and ten-file OSF inventory",
"official Illinois February 2018, July 2018, and February 2019 charts",
"official NCBE 2022 comparison pages"
],
"result": "No captured launch-edition source identifies the UBE chart or interpolation. The pinned Git-body search found no chart identity, interpolation, percentile value, Illinois reference, or comparison-population phrase; README.md contains one generic test-taker occurrence. Katz's later VOR names February 2018 as a proxy and says no July 2022 national percentile distribution was public.",
"status": "unresolved-no-credit",
"limitations": "This is a bounded corpus result, not proof that no private or now-lost source existed."
}
],
"recommendation": {
"author": "GO",
"scope": "Proceed to a later candidate dossier about a real historical score whose percentile is comparison-class-dependent and whose launch derivation remains unresolved.",
"not_authorized": "No dossier admission, public verdict, current-model ranking, lawyer-quality claim, feature, deployment, provider call, or rerun.",
"independent_review": "pending"
},
"limitations": [
"The exact launch chart, UBE comparison population, and interpolation remain unknown and receive no inferred value.",
"No official national July 2022 total-UBE score distribution was captured.",
"Purchased exam questions, proprietary model snapshots, official essay grading, and a complete rerun remain unavailable or unauthorized.",
"The OSF node is private with an anonymous view-only capability and has no confirmed license.",
"The pinned Katz repository has no confirmed license and does not redistribute the purchased MBE exam.",
"Martínez's ranks are modeled from aggregate inputs and assumptions; they are not observed national percentile tables.",
"Historical simulated-exam performance does not establish current model behavior, general legal competence, or practicing-lawyer quality."
]
}
Build receipt
Reproduce this projection
- Catalog
em:catalog:sha256:9bfc972213cba2cde167386103dc2c011ee74639fb7f0794c54120fbbdef1a5d- Frontier
em:frontier:sha256:f33be3eae4c75232d56750ef9a1aa79d96274ece3417d65a75c1391bf61a81bf- Accepted commit
f92846570180dfa4511263f8ba98ecd18f7772c9- Epistemic policy
commons-balanced-v0.1- Disclosure policy
public-noninterference-v0.1- Compiler
epistemedia/0.2.0