<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Hum Factors</journal-id><journal-id journal-id-type="publisher-id">humanfactors</journal-id><journal-id journal-id-type="index">6</journal-id><journal-title>JMIR Human Factors</journal-title><abbrev-journal-title>JMIR Hum Factors</abbrev-journal-title><issn pub-type="epub">2292-9495</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v13i1e95644</article-id><article-id pub-id-type="doi">10.2196/95644</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Sentence-Level Provenance for AI Medical Record Summarization in a Click-to-Inspect Interface: Formative Usability Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Parambath</surname><given-names>Andrew</given-names></name><degrees>MD, MBA, MEd</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Pulpo</surname><given-names>Giordana</given-names></name><degrees>MPS</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hartman</surname><given-names>Vince</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Stanford Medicine</institution><addr-line>900 Welch Road, Suite 350</addr-line><addr-line>Palo Alto</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Abstractive Health</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Kushniruk</surname><given-names>Andre</given-names></name></contrib><contrib contrib-type="editor"><name name-style="western"><surname>Balcarras</surname><given-names>Matthew</given-names></name></contrib><contrib contrib-type="editor"><name name-style="western"><surname>Law</surname><given-names>Stephanie</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Koulas</surname><given-names>Ioannis</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Jonker</surname><given-names>Richard</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Andrew Parambath, MD, MBA, MEd, Stanford Medicine, 900 Welch Road, Suite 350, Palo Alto, CA, 94304, United States, 1 2672979144; <email>andrewparambath@gmail.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>8</day><month>9</month><year>2026</year></pub-date><volume>13</volume><elocation-id>e95644</elocation-id><history><date date-type="received"><day>18</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>27</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Andrew Parambath, Giordana Pulpo, Vince Hartman. Originally published in JMIR Human Factors (<ext-link ext-link-type="uri" xlink:href="https://humanfactors.jmir.org">https://humanfactors.jmir.org</ext-link>), 8.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Human Factors, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://humanfactors.jmir.org">https://humanfactors.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://humanfactors.jmir.org/2026/1/e95644"/><abstract><sec><title>Background</title><p>Large language models can generate fluent summaries of longitudinal medical records, but in high-stakes clinical settings, verification burden remains a barrier to trust. Existing provenance mechanisms such as document-level citations and section references often require manual search within long, fragmented notes, limiting their usefulness during time-constrained workflows for clinicians.</p></sec><sec><title>Objective</title><p>This study aimed to design and evaluate a sentence-level provenance interface (&#x201C;click to inspect&#x201D;) that enables rapid verification of AI-generated longitudinal medical record summaries at the level of individual statements.</p></sec><sec sec-type="methods"><title>Methods</title><p>Between November 2023 and January 2024, we conducted a formative usability study using remotely moderated usability sessions via Zoom to evaluate a web-based sentence-level provenance interface for AI-generated longitudinal medical record summaries. A convenience sample of clinicians was recruited through email outreach to academic and professional networks across the United States. Formative usability testing was conducted with 46 clinician interactions using synthetic longitudinal patient charts. Participants included medical students, residents, and attending physicians across multiple specialties, including internal medicine, dermatology, radiology, plastic surgery, anesthesiology, interventional radiology, obstetrics and gynecology, and family medicine. Usability was assessed using the System Usability Scale and net promoter score, alongside qualitative feedback.</p></sec><sec sec-type="results"><title>Results</title><p>Clinicians reported high usability (mean System Usability Scale score 86.25, SD 7.77; 95% CI 83.96&#x2010;88.54 from 46 participants) and a positive overall experience (net promoter score of 35; 22/46, 47.8% promoters; 18/46, 39.1% passives; and 6/46, 13% detractors). Participants described rapid access to supporting evidence as critical for trust calibration during first-pass chart review. Qualitative feedback identified friction in traditional citation-based interfaces and supported sentence-level inspectability as a low-friction verification mechanism.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Sentence-level provenance transforms AI-generated summaries from static narratives into interactive verification tools. An approach that enables rapid, selective inspection of individual claims during longitudinal chart review may reduce verification burden and support calibrated reliance in high-risk clinical contexts.</p></sec></abstract><kwd-group><kwd>medical informatics</kwd><kwd>usability testing</kwd><kwd>clinical decision support systems</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>natural language processing</kwd><kwd>electronic health records</kwd><kwd>human-centered design</kwd><kwd>health IT</kwd><kwd>trust calibration</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Large language models can generate fluent summaries of long clinical text, yet in high-risk domains such as health care, trust depends not only on accuracy but also on verifiability [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. This is especially true for longitudinal medical record summaries, which condense years of fragmented documentation into a single narrative [<xref ref-type="bibr" rid="ref3">3</xref>]. Unlike encounter summaries written immediately after care [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>], longitudinal summaries often serve as a clinician&#x2019;s first exposure to an unfamiliar patient&#x2019;s history.</p><p>Without prior mental models, clinicians may struggle to distinguish rare but true details from model error and may find it difficult to trust these outputs [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. The burden of verification in time-constrained environments associated with locating supporting evidence within long and redundant notes remains a central barrier to adoption [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Most deployed clinical summarization systems have focused on encounter-based summaries such as visit notes, discharge summaries, and emergency department documentation, which are generated immediately after a clinical interaction [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. In these settings, clinicians already possess a mental model of the patient from direct participation in care, and summaries primarily support recall, documentation, and communication [<xref ref-type="bibr" rid="ref11">11</xref>]. In contrast, longitudinal medical record summaries synthesize years of fragmented documentation across multiple encounters and specialties into a single narrative [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. This creates a distinct cognitive challenge because clinicians must determine whether individual summarized statements are supported by the underlying medical record while simultaneously constructing an understanding of the patient&#x2019;s history. As a result, longitudinal summarization requires not only accurate summaries but also efficient mechanisms for verifying individual clinical claims during first-pass review [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p></sec><sec id="s1-2"><title>Prior Work</title><p>Existing provenance approaches commonly rely on document-level citations, section references, or phrase-level highlights [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. While familiar from academic writing, these mechanisms often require manual search within cited documents. In longitudinal clinical documentation, relevant evidence may be embedded within multipage notes, making targeted verification inefficient.</p><p>Recent AI-assisted clinical information retrieval and summarization systems, including Perplexity, OpenEvidence, and Microsoft Copilot, have incorporated provenance through document-level citations, section references, or highlighted supporting snippets to improve transparency [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Similarly, human-centered explainable AI research has emphasized that explanations should be designed to support user decision-making and integrate naturally into existing workflows rather than simply increasing model transparency [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. However, even when provenance is available, clinicians must still navigate lengthy clinical notes and determine whether the cited evidence adequately supports individual summarized claims, a process that can interrupt reading flow and increase cognitive burden during time-constrained chart review [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Although these approaches improve transparency, little work has specifically examined provenance interfaces designed to support rapid verification of individual clinical statements during longitudinal chart review. This represents an important gap in the literature because clinicians reviewing unfamiliar patients must efficiently distinguish supported clinical information from potential model errors while simultaneously building an understanding of the patient&#x2019;s history, yet few provenance interfaces have been designed specifically to support this verification workflow.</p></sec><sec id="s1-3"><title>Study Objectives</title><p>We designed and evaluated a sentence-level provenance interaction model (&#x201C;click to inspect&#x201D;) that makes every summary sentence directly inspectable in its source context. We conducted formative usability testing to assess whether this interaction reduces friction during verification and supports calibrated trust.</p><p>The objective of this study was to evaluate the usability of a sentence-level provenance interface for AI-generated longitudinal medical record summaries. Specifically, we aimed to (1) evaluate the perceived usability of the interface using the System Usability Scale (SUS); (2) assess clinicians&#x2019; overall experience using the net promoter score (NPS); and (3) characterize clinician perceptions of verification workflow, verification burden, and trust through qualitative usability interviews. Collectively, these aims sought to determine whether sentence-level provenance represents a usable and practical approach for supporting rapid verification of AI-generated longitudinal medical record summaries during clinical chart review.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview of Study Design</title><p>This study was a formative usability evaluation of a web-based sentence-level provenance interface for AI-generated longitudinal medical record summaries. The system was evaluated using synthetic patient charts designed to simulate common longitudinal documentation patterns encountered in clinical practice.</p><p>The purpose of this study was to evaluate the usability of the sentence-level provenance interface rather than the accuracy or clinical performance of the underlying summarization system. The AI-generated summaries used during testing were iteratively refined through structured clinician feedback to ensure that they reflected realistic longitudinal clinical documentation and supported evaluation of the verification workflow. Accordingly, the summaries served as a vehicle for evaluating the provenance interface and were not used to assess the performance of a specific language model or summarization approach.</p></sec><sec id="s2-2"><title>The Tool and Interface</title><p>Participants interacted with a web-based sentence-level provenance interface presenting AI-generated longitudinal medical record summaries alongside a source note viewer. Every sentence within the summary was interactive. When a participant selected a summary sentence, the interface opened the originating clinical note in a side-by-side view, automatically scrolled to the corresponding source sentence, and highlighted the matched text in context. This interaction was designed to support a rapid verification workflow by allowing clinicians to inspect supporting evidence without manually searching through source documentation (<xref ref-type="fig" rid="figure1">Figures 1</xref><xref ref-type="fig" rid="figure2"/>-<xref ref-type="fig" rid="figure3">3</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Click-to-inspect summaries in use. A clinician clicks on a sentence in the longitudinal AI summary (left), and the system opens the originating note (right), automatically scrolling to and highlighting the matched supporting sentence in context.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="humanfactors_v13i1e95644_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>The click-to-inspect loop, where selecting a summary sentence opens the originating note, scrolls to the matched supporting sentence, and highlights it in context for rapid verification and return to reading.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="humanfactors_v13i1e95644_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>The general overview of the workflow that a user would go through when navigating the application.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="humanfactors_v13i1e95644_fig03.png"/></fig><p>Two synthetic patient charts were developed for this study. The first represented an acute emergency department encounter involving an ankle injury, diagnostic imaging, and specialty consultation. The second represented a longitudinal outpatient history spanning multiple years of multispecialty documentation. Both charts were designed to reflect common longitudinal documentation patterns encountered in clinical practice, including fragmented histories, cross-specialty documentation, and varying levels of clinical complexity. Synthetic charts were authored and reviewed by board-certified clinicians to ensure clinical realism while avoiding the use of real patient information.</p><p>Sentence-level provenance links were generated using embedding-based semantic similarity between summary sentences and candidate source note sentences. Sentence embeddings were generated using the transformer-based sentence embedding model all-mpnet-base-v2 (Hugging Face). Before embedding, source clinical notes were converted from their structured representation into plain text and segmented into sentence-level units. Text was initially partitioned using line breaks, runs of 2 or more white space characters, bullet symbols, and dash space separators. Each resulting text segment was trimmed of surrounding white space and further segmented into individual sentences using the Natural Language Toolkit (Team NLTK) Punkt sentence tokenizer. Empty segments were discarded. No abbreviation expansion, spelling correction, or clinical terminology normalization was performed before embedding.</p><p>Cosine similarity was calculated between each summary sentence and all candidate sentences within the originating clinical note. For each summary sentence, the highest-scoring source sentence was selected as the provenance reference. A cosine similarity threshold of 0.70 was used to categorize provenance matches. Summary sentences whose highest similarity score was 0.70 or more were categorized as verified, whereas those with scores below 0.70 were categorized as unverified while remaining clickable within the interface. The threshold was established during iterative internal clinical review of provenance quality to identify a value that consistently distinguished clinically meaningful sentence matches from weaker semantic associations. This represented a clinician-informed calibration process rather than a formally validated statistical threshold.</p><p>The provenance interface used a one-to-one sentence mapping. Each summary sentence was linked to a single originating clinical note and a single highest-scoring source sentence within that note. The underlying summarization pipeline preserved the originating note associated with each generated summary sentence and did not generate individual summary sentences by combining information across multiple source notes. Although a summary sentence could synthesize information from multiple sentences within the same note, the interface intentionally displayed only the highest-scoring supporting sentence. This design was selected to provide a single, consistent point of verification while minimizing visual complexity during chart review. Findings from the usability evaluation supported this approach as participants consistently described a single salient provenance location as intuitive and sufficient for verification during longitudinal chart review.</p></sec><sec id="s2-3"><title>Participant Sampling and Recruiting</title><p>Clinicians were recruited between November 2023 and January 2024 through email outreach to medical students, resident physicians, and attending physicians using academic institutional mailing lists, residency program listservs, and professional networks across multiple health systems in the United States. Recruitment emails described the study as a usability evaluation of a clinical summarization interface and invited interested participants to complete a brief interest form. Participation was voluntary. Eligible participants were required to be actively engaged in clinical training or practice. No additional exclusion criteria were applied.</p></sec><sec id="s2-4"><title>Usability Testing Procedures</title><p>Moderated usability sessions were conducted remotely via Zoom (Zoom Video Communications) between November 2023 and January 2024. Each session lasted approximately 30 minutes and followed a semistructured usability interview guide to ensure consistency across participants (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The guide consisted of standardized clinical scenarios; predefined task-based usability assessments; and open-ended interview questions designed to evaluate clinician workflow, verification behavior, and overall usability of the sentence-level provenance interface.</p><p>During each session, participants interacted with the web-based interface while reviewing AI-generated longitudinal patient summaries. Participants were encouraged to use a think-aloud protocol, verbalizing their reasoning as they reviewed summaries and selectively inspected supporting source material. Moderators used nondirective follow-up prompts to clarify why participants chose to verify particular statements and how the interface influenced their verification process while allowing for flexibility to explore emerging usability issues. Participants were not required to inspect a predetermined number of statements, allowing for observation of natural verification behavior during clinical chart review. All sessions were audio and video recorded and transcribed for subsequent qualitative analysis.</p></sec><sec id="s2-5"><title>Data Collection and Analysis</title><p>Following interaction with the interface, participants completed the SUS and an NPS questionnaire. The SUS is a validated 10-item instrument measuring perceived usability on a scale from 0 to 100. The NPS was assessed using the standard likelihood to recommend question (0-10), with respondents categorized as promoters (9-10), passives (7-8), and detractors (0-6).</p><p>In addition to quantitative usability measures, qualitative data were collected during these remotely moderated usability testing sessions via Zoom using a think-aloud session, in which participants described their verification strategies, perceived workflow friction, trust in AI-generated summaries, and overall experience using the interface during the Zoom session.</p><p>Descriptive statistics were calculated for quantitative usability outcomes. Mean SUS scores, SDs, and 95% CIs were calculated across participants. The NPS was calculated using the standard formula of the percentage of promoters minus the percentage of detractors.</p><p>Audio recordings were transcribed and independently reviewed by 2 researchers. Qualitative data were analyzed using an inductive thematic analysis approach to identify recurring themes related to verification workflow, verification burden, trust calibration, and integration into clinical practice. Themes were iteratively refined through discussion until consensus was achieved.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This study was reviewed and approved by Pearl IRB (Indianapolis, Indiana, United States; 2025-0079) under an expedited review process and was determined to involve minimal risk. All study procedures were conducted in accordance with the approved protocol. Participants provided verbal informed consent before taking part. To protect participant privacy and confidentiality, all usability sessions were audio and video recorded solely for research purposes, securely stored in accordance with the approved study protocol, and analyzed by authorized study personnel. Synthetic patient charts were used throughout the study to ensure that no real patient information or protected health information was accessed or disclosed during usability testing. Participants received a US $25 electronic gift card upon completion of the usability session.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>Recruitment invitations were distributed through multiple overlapping institutional mailing lists, residency program listservs, and professional networks. As a result, the exact number of clinicians who received the invitation could not be determined because individual recipients may have received invitations through more than one distribution channel. During the recruitment period, 107 clinicians expressed interest in participating in the study, of whom 46 (43%) completed a moderated usability session and were included in the final analysis (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of study participants (N=46).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Values</td></tr></thead><tbody><tr><td align="left" valign="top">Age range (y)</td><td align="left" valign="top">20&#x2010;60</td></tr><tr><td align="left" valign="top" colspan="2">Clinical specialty groups represented, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Internal medicine</td><td align="char" char="." valign="top">11 (23.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Surgical specialties</td><td align="char" char="." valign="top">10 (21.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Radiology</td><td align="char" char="." valign="top">4 (8.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Dermatology</td><td align="char" char="." valign="top">3 (6.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neurology</td><td align="char" char="." valign="top">3 (6.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ophthalmology</td><td align="char" char="." valign="top">3 (6.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Emergency medicine</td><td align="char" char="." valign="top">2 (4.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Family medicine</td><td align="char" char="." valign="top">2 (4.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Obstetrics and gynecology</td><td align="char" char="." valign="top">2 (4.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pediatrics</td><td align="char" char="." valign="top">2 (4.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hematology and oncology</td><td align="char" char="." valign="top">2 (4.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Anesthesiology</td><td align="char" char="." valign="top">1 (2.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Psychiatry</td><td align="char" char="." valign="top">1 (2.2)</td></tr></tbody></table></table-wrap><p>Recruitment concluded when participant feedback demonstrated recurring and consistent observations regarding workflow, verification behavior, and overall usability. At that point, the sample size was considered sufficient for the formative usability evaluation because additional recruitment was unlikely to yield substantially new usability insights.</p><p>The mean SUS score was 86.25 (SD 7.77; 95% CI 83.96&#x2010;88.54), indicating high perceived usability of the click-to-inspect interface.</p><p>The NPS was 35, calculated using the standard method (percentage of promoters minus percentage of detractors on the 0&#x2010;10 NPS scale). Of the 46 participants, 22 (47.8%) were classified as promoters (scores of 9&#x2010;10), 18 (39.1%) were classified as passives (scores of 7&#x2010;8), and 6 (13%) were classified as detractors (scores of 0&#x2010;6).</p><p>During the moderated think-aloud sessions, participants frequently described rapid access to supporting evidence as critical to their willingness to rely on the summary during initial review. Clinicians reported that being able to verify individual statements with a single click reduced cognitive load compared with document-level citation systems that require manual searching within lengthy notes. Participants noted that traditional provenance mechanisms often identify the source document but do not eliminate the effort required to locate the exact supporting passage within that document.</p><p>Sentence-level inspectability was consistently described as predictable and low friction. Participants reported that the absence of persistent citation markers or confidence indicators preserved reading flow while maintaining immediate access to supporting evidence when needed. Participants also emphasized that the ability to selectively verify clinically important or unexpected statements aligned well with real-world chart review workflows.</p><p>Representative participant comments included the following:</p><disp-quote><p>Being able to click directly on the sentence and see the exact line in the note makes it much easier to trust the summary.</p></disp-quote><disp-quote><p>Normally I have to scroll through the whole note to find where something came from. This saves a lot of time.</p></disp-quote><disp-quote><p>I like that I can verify the statements that matter without breaking my reading flow.</p></disp-quote></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This formative usability evaluation found that a sentence-level provenance interface for AI-generated longitudinal medical record summaries demonstrated high perceived usability among clinicians, with a mean SUS score of 86.25 (SD 7.77) and an NPS of 35. Consistent with the objectives of this study, clinicians reported that the click-to-inspect interaction reduced the effort required to verify AI-generated clinical statements while preserving reading flow during longitudinal chart review. Participants consistently described rapid access to supporting source material as enabling selective verification of clinically important statements without interrupting their workflow. These findings support the usability of sentence-level provenance as a verification interface independent of the underlying summarization model and suggest that embedding provenance directly within individual summary statements may reduce verification burden while supporting calibrated trust in AI-generated longitudinal summaries.</p><p>Our findings suggest that the value of provenance extends beyond simply identifying the source of summarized information. Participants reported that the interface made it easier to verify individual clinical statements while maintaining focus on understanding the patient&#x2019;s history. Instead of navigating through lengthy clinical notes to locate supporting evidence, clinicians could inspect the source material only when they felt that additional verification was necessary. This workflow appeared to reduce the effort associated with verification and allowed clinicians to remain focused on interpreting the summary rather than searching for supporting documentation.</p><p>These findings are consistent with those of prior work in explainable AI showing that transparency mechanisms are most effective when they support user decision-making within existing workflows rather than simply exposing model outputs [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Existing AI-assisted clinical summarization and information retrieval systems commonly provide document-level citations, section references, or highlighted supporting snippets [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. Although these approaches improve transparency, clinicians must still determine whether the cited material adequately supports the summarized clinical claim while navigating lengthy and fragmented documentation. Our findings suggest that sentence-level provenance may reduce this burden by making supporting evidence immediately accessible at the point where verification is needed.</p><p>The findings are also consistent with principles from human factors research. Clinical reasoning is typically performed at the level of individual findings, diagnoses, medication changes, and clinical events rather than entire documents. Participants consistently reported that they wanted the ability to verify only selected statements rather than every sentence within a summary. Clinicians naturally focused on information that was clinically important, unexpected, or likely to influence decision-making. The interface supported this workflow by allowing participants to selectively inspect supporting evidence without interrupting their review of the remainder of the summary. Participants also noted that the absence of persistent citation markers or confidence scores preserved reading flow while maintaining immediate access to supporting evidence when desired. Together, these observations suggest that minimizing visual complexity while preserving rapid inspectability may represent an important design principle for future AI-assisted clinical documentation systems.</p><p>Several limitations should be considered when interpreting these findings. First, this was a formative usability evaluation conducted using synthetic patient charts. Although synthetic data enable controlled testing without privacy concerns, they may not fully capture the variability, ambiguity, and complexity of real-world longitudinal clinical documentation. Second, although participants represented a broad range of clinical specialties, recruitment took place through academic institutional mailing lists, residency program listservs, and professional networks, which may limit the generalizability of these findings to other clinical settings. Third, this study evaluated perceived usability rather than objective performance. Outcomes such as time to verification, verification accuracy, detection of hallucinated or unsupported statements, and downstream effects on clinical decision-making were not measured. Accordingly, this study cannot determine whether sentence-level provenance improves clinicians&#x2019; ability to identify summarization errors or unsupported clinical claims. Finally, provenance mapping relied on embedding-based semantic similarity, and its robustness across heterogeneous clinical documentation warrants further evaluation.</p><p>Future studies should evaluate sentence-level provenance in real-world clinical environments using production electronic health record data and AI-generated longitudinal summaries. Important outcome measures include verification time, verification accuracy, detection of unsupported or hallucinated statements, changes in information-seeking behavior, and downstream effects on clinical decision-making. Comparative studies should also evaluate sentence-level provenance against conventional document navigation and document-level citation interfaces to determine whether it improves verification efficiency, clinical reasoning, and trust calibration. Additional work is needed to understand how alternative provenance presentation strategies, including confidence scores and prioritization cues, influence verification behavior and clinician reliance on AI-generated summaries.</p><p>This study suggests that sentence-level provenance is a feasible and well-accepted approach for supporting verification of AI-generated longitudinal medical record summaries. As generative AI becomes increasingly integrated into electronic health record workflows, the design of verification interfaces may become as important as improvements in model performance. Supporting efficient, low-friction verification has the potential to improve clinician confidence, promote appropriately calibrated trust, and facilitate the safe integration of AI-generated longitudinal summaries into routine clinical practice.</p></sec><sec id="s4-2"><title>Conclusions</title><p>Sentence-level provenance represents a human-centered approach to improving verifiability in AI-generated longitudinal medical record summaries. By making every summary sentence directly inspectable within its source context, the click-to-inspect interaction embeds verification into the act of reading rather than positioning it as an external search task. In formative usability testing, clinicians described the interface as intuitive and compatible with time-constrained workflows and reported high perceived usability. Although further empirical validation is needed, these findings suggest that designing provenance at the level of individual clinical statements may better align AI summarization tools with real-world clinical practice and support calibrated reliance in high-risk settings.</p></sec></sec></body><back><ack><p>The authors acknowledge Caroline Reiner for contributions to the sentence-matching algorithm and user experience interviews. Generative AI (ChatGPT based on GPT-4; OpenAI) was used solely to assist with spelling and grammar corrections during manuscript preparation. The authors reviewed and edited all AI-assisted revisions and take full responsibility for the final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the US National Science Foundation Small Business Innovation Research Phase II award. The views expressed are those of the authors and do not necessarily reflect the views of the National Science Foundation.</p></sec><sec><title>Data Availability</title><p>Synthetic patient charts were used in this study. The datasets generated or analyzed during this study are not publicly available due to study constraints but are available from the corresponding author on reasonable request. Synthetic patient charts were used in this study. The clinical sentence-matching simulation can also be made available for demonstration upon reasonable request. Due to proprietary software components, the underlying code is not publicly available.</p></sec></notes><fn-group><fn fn-type="con"><p>GP contributed to the conceptualization and design of the study, development of the click-to-inspect interface, data collection, data interpretation, and drafting of the manuscript. AP contributed to manuscript drafting, critical revision for important intellectual content, and interpretation of the findings. VH contributed to the conceptualization of the project, study design, manuscript drafting, and critical revision for important intellectual content. All authors reviewed and approved the final version of the manuscript and met the International Committee of Medical Journal Editors criteria for authorship.</p></fn><fn fn-type="conflict"><p>GP and VH are cofounders of Abstractive Health and hold equity interests in the company. AP also holds an equity interest in Abstractive Health. Because the sentence-level provenance interface evaluated in this study was developed by Abstractive Health, these financial interests could potentially benefit from favorable study findings. The authors designed, conducted, analyzed, and reported the study with this potential conflict disclosed.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">NPS</term><def><p>net promoter score</p></def></def-item><def-item><term id="abb2">SUS</term><def><p>System Usability Scale</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Veen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Uden</surname><given-names>C</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title><source>Nat Med</source><year>2024</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1134</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id><pub-id pub-id-type="medline">38413730</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goodman</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Morgan</surname><given-names>DJ</given-names> </name></person-group><article-title>AI-generated clinical summaries require more than accuracy</article-title><source>JAMA</source><year>2024</year><month>02</month><day>27</day><volume>331</volume><issue>8</issue><fpage>637</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.0555</pub-id><pub-id pub-id-type="medline">38285439</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pivovarov</surname><given-names>R</given-names> </name><name name-style="western"><surname>Elhadad</surname><given-names>N</given-names> </name></person-group><article-title>Automated methods for the summarization of electronic health records</article-title><source>J Am Med Inform Assoc</source><year>2015</year><month>09</month><volume>22</volume><issue>5</issue><fpage>938</fpage><lpage>947</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocv032</pub-id><pub-id pub-id-type="medline">25882031</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hartman</surname><given-names>VC</given-names> </name><name name-style="western"><surname>Bapat</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Weiner</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Navi</surname><given-names>BB</given-names> </name><name name-style="western"><surname>Sholle</surname><given-names>ET</given-names> </name><name name-style="western"><surname>Campion</surname><given-names>TR</given-names>  <suffix>Jr</suffix></name></person-group><article-title>A method to automate the discharge summary hospital course for neurology patients</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>11</month><day>17</day><volume>30</volume><issue>12</issue><fpage>1995</fpage><lpage>2003</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad177</pub-id><pub-id pub-id-type="medline">37639624</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Bains</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluating large language models for drafting emergency department encounter summaries</article-title><source>PLOS Digit Health</source><year>2025</year><month>06</month><volume>4</volume><issue>6</issue><fpage>e0000899</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000899</pub-id><pub-id pub-id-type="medline">40526634</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>M</given-names> </name></person-group><article-title>Human reliance on machine learning models when performance feedback is limited: heuristics and risks</article-title><source>CHI &#x2019;21: Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems</source><year>2021</year><publisher-name>Association for Computing Machinery</publisher-name><pub-id pub-id-type="doi">10.1145/3411764.3445562</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rosenbacke</surname><given-names>R</given-names> </name><name name-style="western"><surname>Melhus</surname><given-names>&#x00C5;</given-names> </name><name name-style="western"><surname>McKee</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stuckler</surname><given-names>D</given-names> </name></person-group><article-title>How explainable artificial intelligence can increase or decrease clinicians&#x2019; trust in AI applications in health care: systematic review</article-title><source>JMIR AI</source><year>2024</year><month>10</month><day>30</day><volume>3</volume><fpage>e53207</fpage><pub-id pub-id-type="doi">10.2196/53207</pub-id><pub-id pub-id-type="medline">39476365</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name><name name-style="western"><surname>Monta&#x00F1;a-Brown</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>13</day><volume>8</volume><issue>1</issue><fpage>274</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id><pub-id pub-id-type="medline">40360677</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silberlust</surname><given-names>J</given-names> </name><name name-style="western"><surname>Solanki</surname><given-names>P</given-names> </name><name name-style="western"><surname>Stevens</surname><given-names>ER</given-names> </name><etal/></person-group><article-title>Artificial intelligence-generated encounter summaries: early insights from ambulatory clinicians at a large academic health system</article-title><source>JAMIA Open</source><year>2025</year><volume>8</volume><issue>5</issue><fpage>ooaf096</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf096</pub-id><pub-id pub-id-type="medline">40904519</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Krishna</surname><given-names>K</given-names> </name><name name-style="western"><surname>Khosla</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bigham</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lipton</surname><given-names>ZC</given-names> </name></person-group><article-title>Generating SOAP notes from doctor-patient conversations using modular summarization techniques</article-title><source>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing</source><year>2021</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>4958</fpage><lpage>4972</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.384</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feblowitz</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>A</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Samal</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sittig</surname><given-names>DF</given-names> </name></person-group><article-title>Summarization of clinical information: a conceptual model</article-title><source>J Biomed Inform</source><year>2011</year><month>08</month><volume>44</volume><issue>4</issue><fpage>688</fpage><lpage>699</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2011.03.008</pub-id><pub-id pub-id-type="medline">21440086</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirsch</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Tanenbaum</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Lipsky Gorman</surname><given-names>S</given-names> </name><etal/></person-group><article-title>HARVEST, a longitudinal patient record summarizer</article-title><source>J Am Med Inform Assoc</source><year>2015</year><month>03</month><volume>22</volume><issue>2</issue><fpage>263</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2014-002945</pub-id><pub-id pub-id-type="medline">25352564</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><source>Perplexity</source><access-date>2026-02-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://perplexity.ai/">https://perplexity.ai/</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><source>OpenEvidence</source><access-date>2026-02-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openevidence.com/">https://openevidence.com/</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><source>Microsoft Copilot</source><access-date>2026-02-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://copilot.microsoft.com/">https://copilot.microsoft.com/</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liao</surname><given-names>QV</given-names> </name><name name-style="western"><surname>Varshney</surname><given-names>KR</given-names> </name></person-group><article-title>Human-centered explainable AI (XAI): from algorithms to user experiences</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 20, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2110.10790</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>QV</given-names> </name><name name-style="western"><surname>Bellamy</surname><given-names>RK</given-names> </name></person-group><article-title>Effect of confidence and explanation on accuracy and trust calibration in AI-assisted decision making</article-title><source>FAT* &#x2019;20: Proceedings of the 2020 Conference on Fairness, Accountability, and Transparency</source><year>2020</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>295</fpage><lpage>305</lpage><pub-id pub-id-type="doi">10.1145/3351095.3372852</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Overhage</surname><given-names>JM</given-names> </name><name name-style="western"><surname>McCallie</surname><given-names>D Jr</given-names> </name></person-group><article-title>Physician time spent using the electronic health record during outpatient encounters: a descriptive study</article-title><source>Ann Intern Med</source><year>2020</year><month>02</month><day>4</day><volume>172</volume><issue>3</issue><fpage>169</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.7326/M18-3684</pub-id><pub-id pub-id-type="medline">31931523</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Downing</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Bates</surname><given-names>DW</given-names> </name><name name-style="western"><surname>Longhurst</surname><given-names>CA</given-names> </name></person-group><article-title>Physician burnout in the electronic health record era: are we ignoring the real cause?</article-title><source>Ann Intern Med</source><year>2018</year><month>07</month><day>3</day><volume>169</volume><issue>1</issue><fpage>50</fpage><lpage>51</lpage><pub-id pub-id-type="doi">10.7326/M18-0139</pub-id><pub-id pub-id-type="medline">29801050</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arndt</surname><given-names>BG</given-names> </name><name name-style="western"><surname>Beasley</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Watkinson</surname><given-names>MD</given-names> </name><etal/></person-group><article-title>Tethered to the EHR: primary care physician workload assessment using EHR event log data and time-motion observations</article-title><source>Ann Fam Med</source><year>2017</year><month>09</month><volume>15</volume><issue>5</issue><fpage>419</fpage><lpage>426</lpage><pub-id pub-id-type="doi">10.1370/afm.2121</pub-id><pub-id pub-id-type="medline">28893811</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kambhamettu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Metaxa</surname><given-names>D</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>K</given-names> </name><name name-style="western"><surname>Head</surname><given-names>A</given-names> </name></person-group><article-title>Explainable notes: examining how to unlock meaning in medical notes with interactivity and artificial intelligence</article-title><source>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</source><year>2024</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1145/3613904.3642573</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Vaughan</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>QV</given-names> </name><name name-style="western"><surname>Lombrozo</surname><given-names>T</given-names> </name><name name-style="western"><surname>Russakovsky</surname><given-names>O</given-names> </name></person-group><article-title>Fostering appropriate reliance on large language models: the role of explanations, sources, and inconsistencies</article-title><source>CHI &#x2019;25: Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems</source><year>2025</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1145/3706598.3714020</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Facciani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Joyce</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Citations and trust in LLM generated responses</article-title><source>Proc AAAI Conf Artif Intell</source><year>2025</year><volume>39</volume><issue>22</issue><fpage>23787</fpage><lpage>23795</lpage><pub-id pub-id-type="doi">10.1609/aaai.v39i22.34550</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hoque</surname><given-names>MN</given-names> </name><name name-style="western"><surname>Mashiat</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ghai</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The HaLLMark effect: supporting provenance and transparent use of large language models in writing with interactive visualization</article-title><source>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</source><year>2024</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1145/3613904.3641895</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kambhamettu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Flores</surname><given-names>J</given-names> </name><name name-style="western"><surname>Head</surname><given-names>A</given-names> </name></person-group><article-title>Traceable texts and their effects: a study of summary-source links in AI-generated summaries</article-title><source>CHI EA &#x2019;25: Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems</source><year>2025</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1145/3706599.3719830</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Interview guide for moderators of usability testing sessions.</p><media xlink:href="humanfactors_v13i1e95644_app1.pdf" xlink:title="PDF File, 8 KB"/></supplementary-material></app-group></back></article>