<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JFR</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id>
      <journal-title>JMIR Formative Research</journal-title>
      <issn pub-type="epub">2561-326X</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v10i1e103345</article-id>
      <article-id pub-id-type="pmid">42628031</article-id>
      <article-id pub-id-type="doi">10.2196/103345</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>A Human-Governed Clinical Informatics Framework for Safe AI-Assisted Mental Health Counseling: Secondary Framework Development and Requirement Mapping Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>MacNeill</surname>
            <given-names>Luke</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wang</surname>
            <given-names>Xiaomeng</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wang</surname>
            <given-names>Zhongyan</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Zhu</surname>
            <given-names>Yi</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Yang</surname>
            <given-names>Mi-Ae</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-8128-5731</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Ha</surname>
            <given-names>Kang-Su</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <address>
            <institution>Department of Psychiatry</institution>
            <institution>Rainbow Hospital</institution>
            <addr-line>122 Gyeongyeol-ro, Seo-gu</addr-line>
            <addr-line>Gwangju, 61924</addr-line>
            <country>Republic of Korea</country>
            <phone>82 627176607</phone>
            <email>ksksha@naver.com</email>
          </address>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-3361-9477</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Science and Technology Convergence</institution>
        <institution>Graduate School</institution>
        <institution>Chosun University</institution>
        <addr-line>Gwangju</addr-line>
        <country>Republic of Korea</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Psychiatry</institution>
        <institution>Rainbow Hospital</institution>
        <addr-line>Gwangju</addr-line>
        <country>Republic of Korea</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>College of Medicine</institution>
        <institution>Chosun University</institution>
        <addr-line>Gwangju</addr-line>
        <country>Republic of Korea</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Kang-Su Ha <email>ksksha@naver.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>21</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>10</volume>
      <elocation-id>e103345</elocation-id>
      <history>
        <date date-type="received">
          <day>2</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>30</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>11</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Mi-Ae Yang, Kang-Su Ha. Originally published in JMIR Formative Research (https://formative.jmir.org), 21.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on https://formative.jmir.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://formative.jmir.org/2026/1/e103345" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Natural language processing and large language model systems are increasingly used to support mental health documentation, screening, and follow-up planning. In counseling contexts, model outputs may influence diagnostic framing, risk recognition, and clinical record content. Static performance metrics and fluent generated summaries are not sufficient to support safe implementation without governance, safety gating, human review, and monitoring.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to develop a human-governed clinical informatics framework for safe AI-assisted mental health counseling and make the formative evidence base and requirement-mapping process traceable.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We conducted a secondary framework development and requirement mapping study using the Korean AI Hub psychological counseling dataset, official data description and use documents, released KLUE-BERT risk prediction model materials, released KoAlpaca summary generation resources, and a deidentified 139-case rule-based summary safety screening audit table derived from the original summary comparison file. Raw counseling transcript text, reference summary full text, and generated summary full text are not included in the manuscript or supplementary materials. We extracted failure modes from documented data and model characteristics, released code and configuration files, documentation-reported model metrics, and rule-based proxy flags. Each failure mode was mapped to safety controls, operational criteria, and deployment-level requirements.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>The official documents described 1661 counseling sessions and 465,474 paragraph-level tokens across depression, anxiety disorder, addiction, and normal control groups. Of the 1661 sessions, the documented split included 1339 (80.6%) training, 173 (10.4%) validation, and 149 (9%) test sessions. The summary generation materials documented 1278 training summaries and 139 test summaries. Documentation-reported model metrics included KLUE-BERT accuracies of 71.43% for depression, 73.53% for anxiety, and 66.67% for addiction and KoAlpaca BERTScore precision, recall, and <italic>F</italic><sub>1</sub>-score values of 62.13%, 59.56%, and 60.80%, respectively. The 139-case screening table contained 77 (55.4%) depression, 31 (22.3%) anxiety, and 31 (22.3%) addiction cases. Rule trigger rates included unsupported content proxy flags in 41% (57/139) of cases, overdiagnostic expression proxy flags in 31.7% (44/139) of cases, medicalized expression proxy flags in 54.7% (76/139) of cases, and any rule-based proxy flag in 91.4% (127/139) of cases. These values are conservative rule trigger rates rather than confirmed clinical error rates. The findings informed a 7-stage workflow, 6 safety control layers, an operational safety gate, a workflow-to-control crosswalk, deployment-level transition criteria, and a constructed high-risk example.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>AI-assisted mental health counseling should be implemented as a governed clinical information workflow rather than as an autonomous diagnostic or documentation pathway. The proposed framework specifies safeguards and validation requirements for future supervised evaluations, but it does not itself establish clinical safety or clinical effectiveness. Prospective simulation, clinician usability testing, patient or client feedback, and independent expert validation remain necessary before routine deployment.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>clinical informatics</kwd>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
        <kwd>mental health</kwd>
        <kwd>counseling</kwd>
        <kwd>natural language processing</kwd>
        <kwd>human-governed workflow</kwd>
        <kwd>decision support</kwd>
        <kwd>data governance</kwd>
        <kwd>patient safety</kwd>
        <kwd>implementation science</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Mental health services increasingly rely on digital tools for documentation, triage, screening, and follow-up [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Counseling encounters contain longitudinal information about mood, anxiety, addiction, interpersonal stress, trauma exposure, self-harm, functional impairment, and protective factors [<xref ref-type="bibr" rid="ref3">3</xref>]. Natural language processing and large language model technologies may help summarize counseling encounters and detect risk-related signals, but they also create safety challenges when outputs are presented as clinically meaningful information [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>].</p>
      <p>Counseling AI is not merely a text processing problem. A model-generated risk label or summary can affect how clinicians understand a client, how a record is written, how follow-up decisions are framed, and how risk is escalated [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. False-positive risk predictions may create unnecessary diagnostic labeling and alert fatigue, whereas false-negative outputs or summaries that omit risk signals may reduce recognition of self-harm, violence, addiction, trauma-related danger, or severe functional deterioration [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Fluent but unsupported summaries may appear clinically authoritative even when they contain hallucinated, overdiagnostic, or poorly calibrated content [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p>
      <p>Clinical AI implementation requires more than model checkpoints, demonstration data, or accuracy scores [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. It requires a sociotechnical workflow that defines who reviews AI output, when output may influence documentation, how risk signals are escalated, how errors are reported, and how model behavior is monitored after deployment [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. For mental health AI, these governance requirements are especially important because language-based outputs may influence stigma, safety planning, diagnostic framing, and trust in care [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref11">11</xref>].</p>
      <p>This study aimed to develop a traceable, human-governed clinical informatics framework for safe AI-assisted mental health counseling. The framework is intended for supervised research and pilot settings in hospitals, counseling centers, digital mental health services, and research teams that evaluate or pilot Korean-language AI-assisted counseling resources. The term “human-governed” is used to describe the intended implementation structure: AI output remains advisory and requires documented human review before it affects records or decisions. The framework was not developed through a prospective human-in-the-loop clinical trial.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design</title>
        <p>We conducted a secondary framework development and requirement mapping study. This study did not develop a new autonomous AI model, prospectively test clinical effectiveness, or use a prospective human-in-the-loop component during framework development. Instead, it translated documented data and model characteristics and rule-triggered summary-screening signals from a Korean-language counseling AI use case into a clinical informatics framework. The study design followed design science logic: problem identification, evidence source selection, failure mode extraction, requirement mapping, artifact construction, and specification of future validation needs.</p>
      </sec>
      <sec>
        <title>Data Source and AI Hub Documentation</title>
        <p>The development context was the publicly released AI Hub psychological counseling dataset [<xref ref-type="bibr" rid="ref12">12</xref>], its official data description [<xref ref-type="bibr" rid="ref13">13</xref>], and its use guideline [<xref ref-type="bibr" rid="ref14">14</xref>]. The official documents described 1661 counseling sessions and 465,474 paragraph-level tokens across depression, anxiety disorder, addiction, and normal control groups. The data structure included transcript text and JSON labeling files. The AI Hub data description states that deidentified MP3 audio and transcribed TXT files were produced during data construction, but MP3 audio files were not publicly released because of privacy protection rules; only TXT files were made available as public source data [<xref ref-type="bibr" rid="ref13">13</xref>]. Of the 1661 sessions, the official guideline documented training, validation, and test splits of 1339 (80.6%), 173 (10.4%), and 149 (9%) sessions, respectively. For the summary generation task, 1278 training summaries and 139 test summaries were documented; normal control cases were excluded from summary generation modeling when symptom, risk, and improvement factors were not applicable.</p>
        <p>The annotation schema contained session-level metadata, diagnostic group labels, summary text, silence and total counseling time, paragraph-level speaker and utterance information, symptom factors, risk factors, symptom change factors, and intervention factors. Examples of safety-relevant fields included “suicidal,” “trauma_experience,” “sleep_disturbance,” “irritability,” “craving,” “withdrawal,” “emotional_regulation,” “social_support,” and “cognitive_restructuring.” These fields were used to characterize the counseling AI use case and define failure modes that could affect risk recognition or documentation quality.</p>
      </sec>
      <sec>
        <title>Reference Model Materials</title>
        <p>Released reference model documentation included KLUE-BERT–based risk prediction materials and KoAlpaca 4-bit summary generation materials [<xref ref-type="bibr" rid="ref15">15</xref>]. The KoAlpaca summary generation resources were treated as released AI Hub reference model materials documented in the same AI Hub reference model package [<xref ref-type="bibr" rid="ref15">15</xref>]. The KLUE-BERT risk prediction model was based on bidirectional encoder representations from transformers (BERT) and Korean Language Understanding Evaluation (KLUE)–related resources [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. The reference model documentation described the KLUE-BERT model as receiving a speaker-marked counseling transcript string and producing a 0 or 1 prediction for depression, anxiety, or addiction risk. The released model folders confirmed disease-specific trained model directories for depression, anxiety, and addiction, each containing model configuration, tokenizer configuration, special token mapping, vocabulary, a training argument file, and model weights in safetensors format. In the inspected depression model configuration, the base model was klue/bert-base, the architecture was CustomBertForSequenceRegression, max_position_embeddings was 512, and tokenizer_config specified a model_max_length of 512.</p>
        <p>The inspected KLUE-BERT code did not provide probability-calibrated raw scores, area under the receiver operating characteristic curve, sensitivity, specificity, or confusion matrix export files. The inference logic was implemented as a regression-style scalar output followed by rounding to an integer in the range of 0 to 3 and conversion to a binary zero-vs-nonzero indicator. Therefore, KLUE-BERT model materials were treated as documentation and code-level traceability inputs rather than independent performance reproduction evidence.</p>
        <p>The summary generation model used KoAlpaca 4-bit resources based on EleutherAI/polyglot-ko-12.8b. The inspected model package included training and inference scripts, low-rank adaptation adapter configuration, tokenizer files, and model configuration files. The low-rank adaptation configuration specified <italic>r</italic> of 8, alpha of 32, dropout of 0.05, and query_key_value as the target module. The inference script used the instruction prompt “다음과 같은 상담기록을 보고 요약서를 작성해주세요” “Please review the following counseling record and write a summary.” and generated summaries from the original text field in bertscore_evaluation.xlsx. The released materials were distributed as Colab-executable notebooks and Python source code resources rather than Docker-based container images.</p>
      </sec>
      <sec>
        <title>Rule-Based Summary Safety Screening Audit Source and Deidentification</title>
        <p>For summary screening traceability, we used a 139-case summary comparison file containing case file names, original counseling transcripts, reference summaries, and generated summaries. In this manuscript, “audit” means a retrospective automated formative screening procedure applied to summary characteristics; it does not mean an independent clinical safety audit or psychiatrist-adjudicated outcome review. Because raw counseling text is sensitive, the original transcript text, reference summary full text, and generated summary full text were removed before reporting. A deidentified rule-based summary safety screening table was created with anonymized case identifiers, diagnostic group labels, summary length indicators, section presence indicators, and rule-based proxy flags. This file is provided as <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The proxy flags were used to support traceability of aggregate screening signals and were not treated as independent clinical expert judgments.</p>
      </sec>
      <sec>
        <title>Proxy Rule Development and Reproducibility</title>
        <p>The proxy rules were implemented by the technical author and reviewed by the clinical coauthor. The clinical coauthor reviewed the operational gate domains, keyword categories, failure mode mapping, and worked example to ensure that the rules were clinically interpretable as conservative screening prompts rather than confirmed clinical errors. The initial screening domains were specified before computing the final aggregate counts by using the AI Hub schema fields, expected summary sections, and known counseling safety concerns. After preliminary inspection, category names were clarified for reporting consistency, but the final rule definitions and aggregate counts were computed from the locked rule set described in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>The screening procedure compared section presence and keyword presence between reference summary text and generated summary text before those texts were removed from the reporting file. The main rule families were (1) required section omission rules for risk, improvement, and intervention sections; (2) high-risk keyword omission rules for self-harm and suicide, trauma and violence, and addiction risk categories; (3) unsupported content proxy rules for critical clinical or risk-related keywords that appeared in generated summaries without support in the original transcript or reference summary; (4) overdiagnostic or medicalized expression rules; and (5) length ratio rules. Generated-to-reference length ratios below 0.50 were classified as too short, and ratios above 1.50 were classified as too long. These thresholds were chosen as conservative heuristics to identify unusually compressed or expanded summaries requiring human review, not as validated clinical thresholds. No formal manual false positive or false negative adjudication study was conducted; this limitation is stated explicitly.</p>
      </sec>
      <sec>
        <title>Failure Mode Identification and Control Mapping</title>
        <p>We identified failure modes from four sources: (1) official data and model documentation; (2) released code, model folders, and execution resource files; (3) documentation-reported model results and the absence of raw score or confusion matrix exports; and (4) rule-based proxy flags from the 139-case screening table. Failure modes were grouped into data and labeling, model, output, human review, privacy, and governance categories. Recognized bias categories from clinical AI literature, including label bias, class imbalance, calibration uncertainty, missing external validation, data leakage concerns, automation bias, and deployment drift, were used as a cross-check so that the failure mode list would not rely solely on author judgment [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p>
        <p>Each failure mode was mapped to a minimum required control using the following rule: if a failure mode could alter risk recognition, diagnostic framing, record content, privacy protection, or accountability, then the framework required a control that prevented automatic use of the AI output, required a documented human decision, or triggered monitoring or governance review. In the workflow-to-control crosswalk, “Required” means that the control is a minimum condition for that workflow stage because it prevents uncontrolled record integration, risk omission, privacy leakage, or accountability gaps. “Support” means that the control reinforces safety but is not the minimum gatekeeping condition for that specific stage.</p>
      </sec>
      <sec>
        <title>Safety Gate and Deployment-Level Criteria</title>
        <p>The safety gate was operationalized as a prerecord integration checkpoint. A generated summary or risk flag may proceed to clinician review only if it has passed source consistency screening, high risk omission screening, overdiagnostic language screening, privacy screening, and required section screening. In routine deployment, source consistency should be assessed against the original counseling source or other clinically available evidence because reference summaries may not be available outside retrospective evaluation. If any gate item is positive, the output remains a draft and must be edited, rejected, escalated, or accompanied by additional assessment before final record integration. High-risk content related to self-harm, suicide, violence, abuse, severe functional deterioration, withdrawal, intoxication, or relapse cannot be silently dismissed; the reviewer must document the rationale for the final decision.</p>
        <p>Deployment levels were defined as transition states rather than descriptive labels. Progression from one level to another requires documented evidence of local safety, review burden, error handling, and governance sign-off. Level 0 is offline research only. Level 1 permits supervised internal pilot-testing with mandatory clinician review. Level 2 permits limited decision support pilot use after local validation and escalation compliance review. Level 3 permits routine supervised use only if postdeployment monitoring demonstrates acceptable safety, workflow burden, and incident response. Level 4 autonomous use is not recommended for AI-assisted mental health counseling under the current evidence base.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>No participants were recruited, no intervention was delivered, and the authors did not contact human participants for the present secondary framework development analysis. The study used released AI Hub documentation, released reference model resources, and an author-generated deidentified rule-based screening table derived from the 139-case summary comparison resource. Raw counseling transcripts, reference summary full text, generated summary full text, and identifiable excerpts were not reproduced in the manuscript or supplementary materials.</p>
        <p>The official AI Hub use guideline for psychological counseling data describes the original data construction process, participant explanation and consent procedures, privacy safeguards, and deidentification procedures for transcript and audio data [<xref ref-type="bibr" rid="ref14">14</xref>]. According to the guideline, textual identifiers were replaced with entity markers, such as “@NAME,” “@ADDRESS,” “@PHONE,” “@DOB,” “@EMAIL,” “@SCHOOL,” and “@HOSPITAL,” and audio data were deidentified through voice transformation, masking, silence insertion, or beep processing [<xref ref-type="bibr" rid="ref14">14</xref>]. The guideline also indicates that consent for personal information use was obtained during the original data construction [<xref ref-type="bibr" rid="ref14">14</xref>].</p>
        <p>Because this secondary data analysis used anonymized and publicly accessible data, it was exempt from approval by the institutional review board under local regulatory policies. The ethical rationale was considered in relation to general human subject research principles [<xref ref-type="bibr" rid="ref18">18</xref>]; the Korean Bioethics and Safety Act definition and review context for human subject research [<xref ref-type="bibr" rid="ref19">19</xref>]; and the consent, privacy, confidentiality, intrusiveness, and potential harm considerations discussed by Eysenbach and Till [<xref ref-type="bibr" rid="ref20">20</xref>] for sensitive online or community research. Before any prospective implementation, local deployment, or patient- or client-facing use, institutional ethics, privacy, and governance review would be required according to local policy.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Evidence Inventory and Traceability</title>
        <p>The framework was derived from a traceable set of documentation, model resource, and deidentified screening inputs. These inputs did not establish prospective clinical safety, but they identified where controls were needed before any counseling AI output could be used in documentation, triage, or follow-up planning. <xref ref-type="table" rid="table1">Table 1</xref> summarizes how each evidence source contributed to a safety signal or framework requirement.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Evidence sources and traceability for framework development.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="180"/>
            <col width="360"/>
            <col width="270"/>
            <col width="190"/>
            <thead>
              <tr valign="top">
                <td>Evidence source</td>
                <td>Traceable item</td>
                <td>Derived signal or design implication</td>
                <td>Framework use</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>AI Hub data description and use guideline</td>
                <td>1661 sessions; 465,474 paragraph-level tokens; depression, anxiety, addiction, and normal groups; JSON schema</td>
                <td>Counseling data contain risk, symptom, change, and intervention fields that can influence risk recognition and documentation</td>
                <td>Data governance, annotation audit, and safety gate requirements</td>
              </tr>
              <tr valign="top">
                <td>Official training, validation, and test documentation</td>
                <td>1339 training, 173 validation, and 149 test sessions; summary task with 1278 training and 139 test summaries</td>
                <td>Model resources are tied to a defined but nonclinical validation structure</td>
                <td>Deployment levels require local validation before clinical use</td>
              </tr>
              <tr valign="top">
                <td>KLUE-BERT released code and model folders</td>
                <td>Disease-specific folders for depression, anxiety, and addiction; configuration and tokenizer files; model_max_length=512 in inspected depression model</td>
                <td>Model identity and execution structure are traceable, but raw score, AUROC<sup>a</sup>, specificity, sensitivity, and confusion matrix exports were not available</td>
                <td>Model governance, version lock, threshold policy, and limitation statement</td>
              </tr>
              <tr valign="top">
                <td>KoAlpaca code and summary generation files</td>
                <td>LoRA<sup>b</sup> adapter, inference script, prompt, generation settings, and bertscore_evaluation.xlsx input</td>
                <td>Generated summaries can be screened for omitted risk information, unsupported content, and unsafe wording</td>
                <td>Output layer safety gate and mandatory human review</td>
              </tr>
              <tr valign="top">
                <td>139-case deidentified rule-based screening table</td>
                <td>77 depression cases, 31 anxiety cases, and 31 addiction cases; full transcript text removed</td>
                <td>Rule-based proxy flags identify outputs requiring structured review; flags are not expert clinical judgments</td>
                <td>Traceable aggregate screening signals and worked example</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>AUROC: area under the receiver operating characteristic curve.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>LoRA: low-rank adaptation.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Dataset and Reference Model Characteristics</title>
        <p>The AI Hub resources documented a balanced clinical use case across depression, anxiety disorder, addiction, and normal control groups. The normal control group was smaller than each clinical group, and the released classification resources reported accuracy values rather than full clinical validation statistics. For this reason, the framework treats model results as decision support signals that require local validation before operational use. <xref ref-type="table" rid="table2">Table 2</xref> summarizes the dataset and reference model characteristics.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Dataset and reference model characteristics.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="390"/>
            <col width="420"/>
            <thead>
              <tr valign="top">
                <td>Domain</td>
                <td>Documentation-confirmed characteristic</td>
                <td>Safety implication</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Dataset composition</td>
                <td>Depression: 484 sessions; anxiety: 487 sessions; addiction: 448 sessions; normal control: 242 sessions</td>
                <td>Local deployment should examine class distribution and alert burden</td>
              </tr>
              <tr valign="top">
                <td>Annotation schema</td>
                <td>Session-level labels, paragraph text, symptom_factor, risk_factor, symptom_change, and intervention_factor</td>
                <td>Label validity and paragraph-level scoring should be audited before treating risk fields as clinical truth</td>
              </tr>
              <tr valign="top">
                <td>Risk prediction model</td>
                <td>KLUE-BERT model for 0 or 1 prediction of depression, anxiety, or addiction</td>
                <td>Outputs should be interpreted as decision support signals, not diagnoses</td>
              </tr>
              <tr valign="top">
                <td>Summary model</td>
                <td>KoAlpaca 4-bit model for structured summary reports</td>
                <td>Generated summaries require source consistency and omission screening</td>
              </tr>
              <tr valign="top">
                <td>Reported classification performance</td>
                <td>Official accuracies: 71.43% for depression, 73.53% for anxiety, 66.67% for addiction, and 70.54% weighted average</td>
                <td>Documentation-reported accuracy alone is insufficient for clinical deployment; sensitivity, specificity, calibration, and subgroup behavior remain required</td>
              </tr>
              <tr valign="top">
                <td>Reported summary performance</td>
                <td>Official BERTScore precision: 62.13%; recall: 59.56%; <italic>F</italic><sub>1</sub>-score: 60.80%</td>
                <td>Semantic similarity does not establish clinical safety; omission and unsupported content checks remain required</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Rule-Based Proxy Criteria</title>
        <p>The proxy criteria were designed to identify outputs that should be reviewed by a human, not label outputs as confirmed clinical errors. <xref ref-type="table" rid="table3">Table 3</xref> summarizes the main rule families and the interpretation of positive flags. Positive flags may reflect conservative screening rules and do not necessarily indicate clinically harmful outputs.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Rule-based proxy criteria and interpretation.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="240"/>
            <col width="450"/>
            <col width="310"/>
            <thead>
              <tr valign="top">
                <td>Rule family</td>
                <td>Operational rule</td>
                <td>Interpretation</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Section omission</td>
                <td>Expected reference summary section is present, but the corresponding generated summary section is absent</td>
                <td>Possible completeness issue requiring structured review</td>
              </tr>
              <tr valign="top">
                <td>High-risk keyword omission</td>
                <td>Reference or source content contains self-harm or suicide, trauma or violence, or addiction risk keywords not reflected in the generated summary</td>
                <td>Possible high risk omission requiring clinician review</td>
              </tr>
              <tr valign="top">
                <td>Unsupported content proxy</td>
                <td>Generated critical clinical or risk keywords are absent from both the reference summary and original transcript</td>
                <td>Possible unsupported content requiring source consistency review</td>
              </tr>
              <tr valign="top">
                <td>Overdiagnostic or medicalized expression proxy</td>
                <td>Generated summary contains diagnostic certainty; disease, treatment, or patient language; or related medicalized terms</td>
                <td>Possible terminology or framing issue requiring editing</td>
              </tr>
              <tr valign="top">
                <td>Length ratio rule</td>
                <td>Generated-to-reference length ratio of &#60;0.50 or &#62;1.50</td>
                <td>Unusually short or long summary requiring completeness and review burden assessment</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>The 139-Case Rule-Based Screening Findings</title>
        <p>The deidentified 139-case screening table retained only anonymized identifiers, diagnostic group labels, summary length indicators, section presence indicators, and rule-based proxy flags. It did not retain raw transcript text, reference summary text, or generated summary text. The group distribution was 55.4% (n=77) depression cases, 22.3% (n=31) anxiety cases, and 22.3% (n=31) addiction cases.</p>
        <p>Rule trigger rates showed that generated summaries frequently required structured review under conservative screening rules. Section-level omissions were less frequent than language or content proxy flags: risk factor section omission occurred in 2.9% (4/139) of cases, improvement factor omission occurred in 8.6% (12/139) of cases, and intervention factor omission occurred in 13.7% (19/139) of cases. High risk omission proxy flags were observed for self-harm and suicide keywords in 5.8% (8/139) of cases, trauma and violence keywords in 15.1% (21/139) of cases, and addiction risk keywords in 13.7% (19/139) of cases. Unsupported content proxy flags occurred in 41% (57/139) of cases, overdiagnostic expression proxy flags occurred in 31.7% (44/139) of cases, and medicalized expression proxy flags occurred in 54.7% (76/139) of cases. These results show why human review and source consistency checking are needed, but they should not be interpreted as confirmed hallucination rates, clinical error rates, or unsafe summary rates. <xref ref-type="table" rid="table4">Table 4</xref> summarizes the aggregate rule-based summary safety-screening proxy flags by diagnostic group.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Aggregate rule-based summary safety screening proxy flags. Positive flags reflect conservative automated rule triggers and do not necessarily indicate clinically harmful outputs or independently adjudicated clinical errors.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="120"/>
            <col width="150"/>
            <col width="110"/>
            <col width="140"/>
            <col width="270"/>
            <thead>
              <tr valign="top">
                <td>Rule-based proxy flag</td>
                <td>Overall count, n/N (%)</td>
                <td>Depression, n/N (%)</td>
                <td>Anxiety, n/N (%)</td>
                <td>Addiction, n/N (%)</td>
                <td>Interpretation for framework design</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Risk factor section omission</td>
                <td>4/139 (2.9)</td>
                <td>2/4 (50)</td>
                <td>1/4 (25)</td>
                <td>1/4 (25)</td>
                <td>Generated summaries require a structured risk factor section check before record integration</td>
              </tr>
              <tr valign="top">
                <td>Improvement factor section omission</td>
                <td>12/139 (8.6)</td>
                <td>4/12 (33.3)</td>
                <td>4/12 (33.3)</td>
                <td>4/12 (33.3)</td>
                <td>Generated summaries may omit recovery or protective content relevant to follow-up planning</td>
              </tr>
              <tr valign="top">
                <td>Intervention factor section omission</td>
                <td>19/139 (13.7)</td>
                <td>7/19 (36.8)</td>
                <td>5/19 (26.3)</td>
                <td>7/19 (36.8)</td>
                <td>Counselor intervention content should be checked before summaries are used for continuity of care</td>
              </tr>
              <tr valign="top">
                <td>Self-harm or suicide keyword omission</td>
                <td>8/139 (5.8)</td>
                <td>5/8 (62.5)</td>
                <td>2/8 (25)</td>
                <td>1/8 (12.5)</td>
                <td>Possible omission requires mandatory human review and escalation logic</td>
              </tr>
              <tr valign="top">
                <td>Trauma or violence keyword omission</td>
                <td>21/139 (15.1)</td>
                <td>15/21 (71.4)</td>
                <td>6/21 (28.6)</td>
                <td>0/21 (0)</td>
                <td>Trauma- or violence-related content requires high-risk checklist screening</td>
              </tr>
              <tr valign="top">
                <td>Addiction risk keyword omission</td>
                <td>19/139 (13.7)</td>
                <td>10/19 (52.6)</td>
                <td>6/19 (31.6)</td>
                <td>3/19 (15.8)</td>
                <td>Addiction-related risk content requires relapse, withdrawal, or intoxication screening</td>
              </tr>
              <tr valign="top">
                <td>Unsupported content proxy</td>
                <td>57/139 (41)</td>
                <td>34/57 (59.6)</td>
                <td>15/57 (26.3)</td>
                <td>8/57 (14)</td>
                <td>Fluent generated text requires source consistency review before it is trusted</td>
              </tr>
              <tr valign="top">
                <td>Overdiagnostic expression proxy</td>
                <td>44/139 (31.7)</td>
                <td>27/44 (61.4)</td>
                <td>9/44 (20.5)</td>
                <td>8/44 (18.2)</td>
                <td>Diagnostic certainty and stigmatizing language require terminology guardrails</td>
              </tr>
              <tr valign="top">
                <td>Medicalized expression proxy</td>
                <td>76/139 (54.7)</td>
                <td>42/76 (55.3)</td>
                <td>15/76 (19.7)</td>
                <td>19/76 (25)</td>
                <td>Medicalized wording should be reviewed to avoid inappropriate clinical framing</td>
              </tr>
              <tr valign="top">
                <td>Generated summary too short</td>
                <td>34/139 (24.5)</td>
                <td>15/34 (44.1)</td>
                <td>8/34 (23.5)</td>
                <td>11/34 (32.4)</td>
                <td>Abnormally short summaries require completeness review</td>
              </tr>
              <tr valign="top">
                <td>Generated summary too long</td>
                <td>5/139 (3.6)</td>
                <td>3/5 (60)</td>
                <td>1/5 (20)</td>
                <td>1/5 (20)</td>
                <td>Overly long summaries may increase review burden and require editing</td>
              </tr>
              <tr valign="top">
                <td>Any rule-based proxy flag</td>
                <td>127/139 (91.4)</td>
                <td>75/127 (59.1)</td>
                <td>25/127 (19.7)</td>
                <td>27/127 (21.3)</td>
                <td>Most draft outputs triggered at least one conservative review rule</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Failure Modes and Required Controls</title>
        <p>The identified failure modes show how a documentation-level AI resource can create implementation risks if outputs are treated as clinically authoritative. The corresponding controls are designed to keep generated summaries and risk outputs in a draft or decision support state until a responsible human reviewer has assessed them. <xref ref-type="table" rid="table5">Table 5</xref> summarizes the failure mode–to-control mapping.</p>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Failure modes and required controls.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="230"/>
            <col width="190"/>
            <col width="160"/>
            <col width="230"/>
            <thead>
              <tr valign="top">
                <td>Failure mode</td>
                <td>Evidence source</td>
                <td>Potential harm</td>
                <td>Required control</td>
                <td>Minimum operational criterion</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Label and annotation uncertainty</td>
                <td>Official schema and paragraph-level scoring; nonexpert and expert review documented, but independent reliability not available</td>
                <td>Risk labels may inherit annotator assumptions or class distribution effects</td>
                <td>Label audit control</td>
                <td>Before pilot use, document class distribution, missingness, and interrater or expert review process for local labels</td>
              </tr>
              <tr valign="top">
                <td>Positive or poorly calibrated risk prediction</td>
                <td>Official accuracy only; no raw score, sensitivity, specificity, calibration, or confusion matrix exports available</td>
                <td>Unnecessary labeling, alert fatigue, and missed false negative analysis</td>
                <td>Model governance control</td>
                <td>Local validation must report sensitivity, specificity, calibration, false positive rate, false negative rate, and review burden before level 2 use</td>
              </tr>
              <tr valign="top">
                <td>Omitted high-risk information</td>
                <td>139-case rule trigger proxy flags</td>
                <td>Self-harm, trauma, violence, addiction, or deterioration may be missed</td>
                <td>Safety gate checklist</td>
                <td>Any high-risk keyword or missing risk section triggers mandatory clinician review and documented disposition</td>
              </tr>
              <tr valign="top">
                <td>Unsupported or hallucinated summary content</td>
                <td>Unsupported content proxy and official hallucination prevention documentation</td>
                <td>Unsupported content may enter clinical record</td>
                <td>Source consistency control</td>
                <td>Reviewer must compare AI draft against the original source or clinically available source evidence before approval</td>
              </tr>
              <tr valign="top">
                <td>Overdiagnostic or medicalized wording</td>
                <td>Overdiagnostic and medicalized expression proxy flags</td>
                <td>Stigma, premature diagnosis, and inappropriate referral or treatment framing</td>
                <td>Terminology guardrail</td>
                <td>AI output must avoid definitive diagnosis unless confirmed by a qualified clinician; uncertain language must be edited</td>
              </tr>
              <tr valign="top">
                <td>Automation bias</td>
                <td>Known clinical AI risk and framework use case</td>
                <td>Clinicians may overtrust fluent text</td>
                <td>Interface and human review control</td>
                <td>Output displayed as draft; final record requires active human confirmation, edit, or rejection</td>
              </tr>
              <tr valign="top">
                <td>Privacy leakage</td>
                <td>AI Hub deidentification rules and data use restrictions</td>
                <td>Sensitive counseling information may be disclosed or redistributed</td>
                <td>Data governance control</td>
                <td>Raw transcripts not redistributed; role-based access and deidentification retained</td>
              </tr>
              <tr valign="top">
                <td>Accountability gap</td>
                <td>Need for record integration and incident response</td>
                <td>Unclear responsibility after AI-related error</td>
                <td>Governance control</td>
                <td>Named clinical owner, audit trail, incident pathway, and model version record required</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Proposed Human-Governed Framework</title>
        <p>The framework consists of seven workflow stages: (1) data intake and deidentification, (2) AI service execution, (3) structured safety gating, (4) clinician review, (5) final record integration, (6) postdeployment monitoring, and (7) institutional governance. These stages are paired with 6 control layers: governance, data, model, output, human review, and postdeployment surveillance. AI output is always treated as draft or decision support content. It is not a final diagnosis, final counseling note, or autonomous escalation decision. The framework development workflow is shown in <xref rid="figure1" ref-type="fig">Figure 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Human-governed framework development workflow. The figure shows how AI Hub documentation, model resources, and the 139-case deidentified rule-based summary safety screening audit were converted into failure modes, requirements, and a human-governed counseling AI implementation framework.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103345_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p><xref rid="figure2" ref-type="fig">Figure 2</xref> operationalizes the framework as a 4-step prerecord safety pathway. AI-generated risk predictions or summaries initially remain as draft suggestions and are screened for high-risk omissions, unsupported content, overdiagnostic wording, privacy concerns, and required section completeness. A clinician must then approve, edit, reject, or request additional assessment, and only human-approved content may proceed to record integration and postdeployment monitoring.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Operational safety gate for AI-assisted counseling output. AI-generated outputs remain as draft suggestions until source consistency, high risk omission, overdiagnostic wording, privacy, and required section checks have been completed and a human decision has been documented.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103345_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p><xref rid="figure3" ref-type="fig">Figure 3</xref> shows how the 6 safety control layers are distributed across the 7 workflow stages rather than applied as isolated checklists. Required cells identify the minimum gatekeeping control for each stage, whereas support cells indicate indirect or reinforcing controls. The crosswalk clarifies where governance, data, model execution, output, human review, and postdeployment responsibilities become mandatory throughout the implementation pathway.</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Workflow stage–to–control layer crosswalk. “Required” indicates a minimum control for that workflow stage because it prevents uncontrolled record integration, risk omission, privacy leakage, or accountability gaps. “Support” indicates an indirect or reinforcing control. The crosswalk links workflow stages to safety control layers so that governance, data, model execution, output, human review, and postdeployment controls are not treated as independent checklists.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103345_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Operational Safety Gate</title>
        <p>The operational safety gate domains, pass criteria, trigger criteria, and required actions are summarized in <xref ref-type="table" rid="table6">Table 6</xref>. The gate is intentionally conservative: a positive proxy flag should lead to structured review, not automatic rejection or automatic clinical escalation. Human reviewers remain responsible for final interpretation and for applying local crisis or clinical protocols.</p>
        <table-wrap position="float" id="table6">
          <label>Table 6</label>
          <caption>
            <p>Operational safety gate criteria.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="300"/>
            <col width="260"/>
            <col width="250"/>
            <thead>
              <tr valign="top">
                <td>Gate domain</td>
                <td>Pass criterion</td>
                <td>Fail or trigger criterion</td>
                <td>Required action</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Source consistency</td>
                <td>Generated summary claims can be supported by the original source or other clinically available evidence</td>
                <td>New diagnostic claims, unsupported facts, or inconsistent content appear</td>
                <td>Edit or reject AI output; document reason</td>
              </tr>
              <tr valign="top">
                <td>Self-harm or suicide</td>
                <td>No self-harm or suicide signal is present, or any signal is accurately represented</td>
                <td>Source content suggests self-harm or suicide, but the generated output omits or minimizes it</td>
                <td>Immediate clinician review; escalation according to local crisis protocol</td>
              </tr>
              <tr valign="top">
                <td>Violence, abuse, or trauma</td>
                <td>Relevant content is accurately represented when present</td>
                <td>Trauma, violence, abuse, or severe threat content is omitted or softened</td>
                <td>Clinician review and risk documentation required</td>
              </tr>
              <tr valign="top">
                <td>Addiction risk</td>
                <td>Relapse, withdrawal, intoxication, craving, or loss of control is represented when present</td>
                <td>Addiction risk content is omitted, minimized, or reframed incorrectly</td>
                <td>Addiction risk review; assess need for follow-up or escalation</td>
              </tr>
              <tr valign="top">
                <td>Overdiagnosis or medicalization</td>
                <td>Output uses cautious, descriptive language</td>
                <td>Output states or implies a diagnosis without clinician confirmation or uses stigmatizing certainty</td>
                <td>Terminology editing or rejection before record integration</td>
              </tr>
              <tr valign="top">
                <td>Privacy</td>
                <td>No identifiers or sensitive redistributable text appear</td>
                <td>Identifier or raw transcript text appears in output or logs beyond approved use</td>
                <td>Remove content, report incident if required, and block record integration</td>
              </tr>
              <tr valign="top">
                <td>Completeness</td>
                <td>Required sections are present when applicable: symptoms, risk factors, improvement factors, and intervention factors</td>
                <td>Missing required section or abnormal summary length</td>
                <td>Manual completeness review before approval</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Deployment-Level Transition Criteria</title>
        <p>The deployment-level transition criteria and authorization requirements are summarized in <xref ref-type="table" rid="table7">Table 7</xref>. The levels are intended to prevent premature clinical use. Higher levels require local validation, escalation compliance, monitoring, and governance sign-off; autonomous use is not authorized under the current evidence base.</p>
        <table-wrap position="float" id="table7">
          <label>Table 7</label>
          <caption>
            <p>Deployment-level transition criteria.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="170"/>
            <col width="240"/>
            <col width="310"/>
            <col width="280"/>
            <thead>
              <tr valign="top">
                <td>Level</td>
                <td>Permitted use</td>
                <td>Required evidence before entering level</td>
                <td>Authorization and stop rule</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Level 0: offline research only</td>
                <td>Retrospective analysis, sandbox testing, and nonclinical framework development</td>
                <td>Data use permission; no patient- or client-facing output; no record integration</td>
                <td>Study lead authorization; stop if raw identifiers or restricted data are exposed</td>
              </tr>
              <tr valign="top">
                <td>Level 1: supervised internal pilot</td>
                <td>Clinician-reviewed draft summaries or safety check support in a controlled internal setting</td>
                <td>Model version locked; safety gate configured; reviewer training completed; audit logging enabled</td>
                <td>Clinical owner and governance lead sign-off; stop if high risk omissions or unsupported content recur</td>
              </tr>
              <tr valign="top">
                <td>Level 2: limited decision support pilot</td>
                <td>Restricted triage or documentation support with mandatory clinician confirmation</td>
                <td>Local validation reports sensitivity, specificity, false positive and false negative rates, calibration, summary review burden, and escalation compliance</td>
                <td>Institutional AI or clinical governance sign-off; suspend if safety gate failures exceed local tolerance or escalation compliance is incomplete</td>
              </tr>
              <tr valign="top">
                <td>Level 3: routine supervised use</td>
                <td>Routine supervised documentation or decision support use with monitoring</td>
                <td>Postdeployment monitoring shows stable performance, acceptable review burden, error resolution process, user training, and periodic revalidation</td>
                <td>Formal institutional approval; stop if drift, unsafe output pattern, or incident review indicates unacceptable risk</td>
              </tr>
              <tr valign="top">
                <td>Level 4: autonomous use</td>
                <td>AI output directly affects records or decisions without human confirmation</td>
                <td>Not recommended for AI-assisted mental health counseling under current evidence</td>
                <td>Not authorized</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Constructed Worked Example: Possible Self-Harm Risk Omission</title>
        <p>The following worked example is a constructed illustrative example based on a generalized high-risk pattern, not a direct quotation from any of the 139 cases and not a modified identifiable case. A counseling source contains indirect references to hopelessness, inability to continue, and family burden. The AI-generated summary describes anxiety and sleep disturbance but does not mention self-harm–related concern or hopelessness. Under the proposed framework, the safety gate identifies a self-harm or suicide omission proxy because clinically available source evidence contains high-risk language not reflected in the generated summary. The system prevents automatic record integration, displays the output as a draft, and requires clinician review. The clinician reviews the source evidence, edits the summary to include a nonstigmatizing risk statement when appropriate, documents a brief rationale, and follows local crisis escalation protocols if immediate risk is suspected. The final record stores only the human-approved summary, the safety gate flag, the reviewer identity, the decision time stamp, and the follow-up action. The case is also logged for postdeployment monitoring of recurring omission patterns.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>This study developed a human-governed clinical informatics framework for AI-assisted mental health counseling. The main finding is that counseling AI should be governed as a clinical information system rather than judged only by model metrics. By making the evidence base traceable, the study showed how AI Hub documentation, released model materials, and a 139-case deidentified rule-based summary safety screening table can be converted into workflow controls, safety gate criteria, deployment levels, and monitoring requirements.</p>
      </sec>
      <sec>
        <title>Contribution Beyond Existing AI Governance Guidance</title>
        <p>The framework is consistent with general health AI governance, clinical prediction reporting, implementation guidance, and clinical decision support literature [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Its specific contribution is the counseling-focused operationalization of those principles. First, it treats generated counseling summaries as draft clinical information that must be checked for risk omission, overdiagnostic wording, and unsupported content before record integration. Second, it includes a safety gate tailored to mental health risks such as self-harm, trauma, violence, addiction, and severe functional decline. Third, it links model governance to workflow decisions by requiring local sensitivity, specificity, calibration, error burden, and escalation compliance evidence before higher deployment levels. The framework also reflects the need to address algorithmic bias and distributional effects before AI outputs influence care pathways [<xref ref-type="bibr" rid="ref23">23</xref>]. Fourth, it specifies that autonomous use is not recommended for AI-assisted mental health counseling under the current evidence base.</p>
      </sec>
      <sec>
        <title>Implications for Hospitals and Counseling Centers</title>
        <p>Institutions considering AI-assisted mental health counseling should begin with level 0 or level 1 use. In practice, this means offline research or supervised internal pilots in which AI-generated summaries and risk predictions are clearly labeled as drafts. Staff should be trained to recognize automation bias and apply the high-risk checklist consistently. Before any system is used for routine triage or documentation support, local leaders should define thresholds, escalation rules, audit responsibilities, privacy procedures, user training, and stopping rules.</p>
        <p>In supervised pilot settings, institutions should disclose to patients or clients when AI-generated draft summaries or risk support tools are used in documentation, triage, or the counseling workflow while clarifying that final interpretation and decisions remain the responsibility of a qualified human professional. At minimum, AI disclosure should be required when AI output influences the clinical record, triage workflow, or client-facing communication.</p>
      </sec>
      <sec>
        <title>Implications for Developers</title>
        <p>Developers should design counseling AI interfaces around clinician control rather than maximum automation. Useful features include provenance metadata, model version display, source-linked summary review, high-risk checklist prompts, edit history, reject and escalation options, and dashboards for error monitoring. Risk scores or generated summaries should not be presented as diagnostic conclusions unless they have been clinically validated for that specific purpose and setting.</p>
        <p>Model layer governance should also consider interpretability and model complexity. When 2 systems provide comparable local safety performance, the more interpretable or easier-to-audit system should be preferred for mental health counseling workflows because it reduces the verification burden placed on human reviewers and supports accountability.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>This study has several limitations. First, it was a secondary framework development study, not a prospective implementation trial. Second, no clinician usability study, patient or client feedback study, expert Delphi panel, or independent external validation was conducted. Third, the framework was developed from a Korean-language AI Hub counseling use case and may require adaptation for other languages, populations, or service settings. Fourth, KLUE-BERT raw prediction scores, area under the receiver operating characteristic curve, sensitivity, specificity, calibration curves, and confusion matrix exports were not available in the released materials reviewed; therefore, the manuscript reports documentation-level model characteristics rather than independently reproduced risk prediction performance. Fifth, the rule-based proxy flags were automated screening triggers intended for traceability and structured review; they were not psychiatrist-adjudicated clinical safety outcomes and should not be interpreted as clinical error or unsafe output rates. Sixth, no formal manual false positive or false negative adjudication study was performed for the proxy flags. Seventh, participant compensation details from the original AI Hub data collection were not available in the released documents reviewed. Eighth, local legal, ethical, and institutional requirements may differ and should be reviewed before any deployment.</p>
      </sec>
      <sec>
        <title>Future Research</title>
        <p>Future studies should prospectively evaluate the framework in simulated and real counseling workflows. Priority outcomes include clinician review time, high-risk information recall, false alert burden, documentation quality, summary correction rate, escalation compliance, incident frequency, calibration drift, patient or client acceptability, and clinician workload. Future work should incorporate independent expert review of rule-based proxy flags and local validation of risk prediction models using sensitivity, specificity, calibration, subgroup performance, and decision curve analysis.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>AI-assisted mental health counseling should be implemented through clinician-governed workflows rather than autonomous documentation or diagnostic pathways. The proposed framework connects data governance, model control, operational safety gating, clinician review, record integration, monitoring, and institutional accountability. The framework specifies safeguards and validation requirements for future supervised evaluation, but it does not itself establish clinical safety or clinical effectiveness.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Deidentified 139-case rule-based summary safety screening flag table derived from bertscore_evaluation.xlsx. Raw counseling transcript text, reference summary text, and generated summary text were removed.</p>
        <media xlink:href="formative_v10i1e103345_app1.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 37 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BERT</term>
          <def>
            <p>bidirectional encoder representations from transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">KLUE</term>
          <def>
            <p>Korean Language Understanding Evaluation</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors acknowledge the public AI Hub psychological counseling data and reference model resources that supported the development of this framework. This research used datasets and documentation from the Open AI Dataset Project (AI Hub, Republic of Korea). During the preparation of this manuscript, OpenAI ChatGPT was used for language drafting and editing assistance. The authors reviewed, verified, and edited all generated content, references, claims, tables, and interpretations and take full responsibility for the accuracy, integrity, and final content of the submitted manuscript, consistent with publication and authorship responsibility recommendations [<xref ref-type="bibr" rid="ref24">24</xref>].</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The underlying psychological counseling dataset and reference model resources are available through the AI Hub subject to its data use procedures and restrictions. The authors cannot redistribute raw counseling transcripts, original labeling files, audio files, or original AI Hub data because they may contain sensitive mental health information and are subject to AI Hub data use conditions. A deidentified, author-generated rule-based summary safety screening table that does not contain raw counseling text, reference summary text, and generated summary text is provided as <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The authors received no direct funding for the preparation of this manuscript. The AI Hub dataset was used as an existing public research resource and did not constitute direct funding to the authors.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>MAY contributed to conceptualization, data curation, formal analysis, investigation, methodology, software, visualization, writing—original draft, and writing—review and editing. KSH contributed to conceptualization, clinical interpretation, investigation, methodology, supervision, validation, writing—review and editing, and corresponding author responsibilities. KSH reviewed the operational gate domains, keyword categories, failure mode mapping, and worked example from a psychiatric and clinical safety perspective. Both authors reviewed and approved the final manuscript and agree to be accountable for all aspects of the work.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Challen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Denny</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pitt</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gompels</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Edwards</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tsaneva-Atanasova</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence, bias and clinical safety</article-title>
          <source>BMJ Qual Saf</source>
          <year>2019</year>
          <month>03</month>
          <volume>28</volume>
          <issue>3</issue>
          <fpage>231</fpage>
          <lpage>7</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://qualitysafety.bmj.com/lookup/pmidlookup?view=long&#38;pmid=30636200"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmjqs-2018-008370</pub-id>
          <pub-id pub-id-type="medline">30636200</pub-id>
          <pub-id pub-id-type="pii">bmjqs-2018-008370</pub-id>
          <pub-id pub-id-type="pmcid">PMC6560460</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Suleyman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>King</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title>
          <source>BMC Med</source>
          <year>2019</year>
          <month>10</month>
          <day>29</day>
          <volume>17</volume>
          <issue>1</issue>
          <fpage>195</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedicine.biomedcentral.com/articles/10.1186/s12916-019-1426-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id>
          <pub-id pub-id-type="medline">31665002</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12916-019-1426-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC6821018</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="web">
          <article-title>Ethics and governance of artificial intelligence for health: WHO guidance</article-title>
          <source>World Health Organization</source>
          <year>2021</year>
          <month>6</month>
          <day>28</day>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.who.int/publications/i/item/9789240029200">https://www.who.int/publications/i/item/9789240029200</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="web">
          <article-title>Artificial Intelligence Risk Management Framework (AI RMF 1.0)</article-title>
          <source>National Institute of Standards and Technology</source>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.6028/NIST.AI.100-1">https://doi.org/10.6028/NIST.AI.100-1</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vasey</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Nagendran</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Campbell</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Clifton</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Denaxas</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Denniston</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Faes</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Geerts</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ibrahim</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Mateen</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>McCradden</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Morgan</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ordish</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Rogers</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Saria</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Watkinson</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Weber</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wheatstone</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>McCulloch</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title>
          <source>Nat Med</source>
          <year>2022</year>
          <month>05</month>
          <volume>28</volume>
          <issue>5</issue>
          <fpage>924</fpage>
          <lpage>33</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id>
          <pub-id pub-id-type="medline">35585198</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-022-01772-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KG</given-names>
            </name>
            <name name-style="western">
              <surname>Dhiman</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Riley</surname>
              <given-names>RD</given-names>
            </name>
            <name name-style="western">
              <surname>Beam</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Van Calster</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ghassemi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Reitsma</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>van Smeden</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Boulesteix</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Camaradou</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Denaxas</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Denniston</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Glocker</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Golub</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Harvey</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Heinze</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Hoffman</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Kengne</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Lam</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Loder</surname>
              <given-names>EW</given-names>
            </name>
            <name name-style="western">
              <surname>Maier-Hein</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Mateen</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>McCradden</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Oakden-Rayner</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ordish</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Parnell</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Rose</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wynants</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Logullo</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title>
          <source>BMJ</source>
          <year>2024</year>
          <month>04</month>
          <day>16</day>
          <volume>385</volume>
          <fpage>e078378</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&#38;pmid=38626948"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj-2023-078378</pub-id>
          <pub-id pub-id-type="medline">38626948</pub-id>
          <pub-id pub-id-type="pmcid">PMC11019967</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Van Calster</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>McLernon</surname>
              <given-names>DJ</given-names>
            </name>
            <name name-style="western">
              <surname>van Smeden</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wynants</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Steyerberg</surname>
              <given-names>EW</given-names>
            </name>
            <collab>Topic Group ‘Evaluating diagnostic tests and prediction models’ of the STRATOS initiative</collab>
          </person-group>
          <article-title>Calibration: the Achilles heel of predictive analytics</article-title>
          <source>BMC Med</source>
          <year>2019</year>
          <month>12</month>
          <day>16</day>
          <volume>17</volume>
          <issue>1</issue>
          <fpage>230</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedicine.biomedcentral.com/articles/10.1186/s12916-019-1466-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12916-019-1466-7</pub-id>
          <pub-id pub-id-type="medline">31842878</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12916-019-1466-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC6912996</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cross</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Choma</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Onofrey</surname>
              <given-names>JA</given-names>
            </name>
          </person-group>
          <article-title>Bias in medical AI: implications for clinical decision-making</article-title>
          <source>PLOS Digit Health</source>
          <year>2024</year>
          <month>11</month>
          <day>7</day>
          <volume>3</volume>
          <issue>11</issue>
          <fpage>e0000651</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000651"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000651</pub-id>
          <pub-id pub-id-type="medline">39509461</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-24-00171</pub-id>
          <pub-id pub-id-type="pmcid">PMC11542778</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tian</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Machine learning approaches for depression detection on social media: a systematic review of biases and methodological challenges</article-title>
          <source>J Behav Data Sci</source>
          <year>2025</year>
          <month>02</month>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>67</fpage>
          <lpage>102</lpage>
          <pub-id pub-id-type="doi">10.35566/jbds/caoyc</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Amann</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Blasimme</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Vayena</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Frey</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Madai</surname>
              <given-names>VI</given-names>
            </name>
            <collab>Precise4Q consortium</collab>
          </person-group>
          <article-title>Explainability for artificial intelligence in healthcare: a multidisciplinary perspective</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2020</year>
          <month>11</month>
          <day>30</day>
          <volume>20</volume>
          <issue>1</issue>
          <fpage>310</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-020-01332-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-020-01332-6</pub-id>
          <pub-id pub-id-type="medline">33256715</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-020-01332-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC7706019</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>McCradden</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Anderson</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>A Stephenson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Drysdale</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Erdman</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Goldenberg</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Zlotnik Shaul</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>A research ethics framework for the clinical translation of healthcare machine learning</article-title>
          <source>Am J Bioeth</source>
          <year>2022</year>
          <month>05</month>
          <volume>22</volume>
          <issue>5</issue>
          <fpage>8</fpage>
          <lpage>22</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/10.1080/15265161.2021.2013977?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1080/15265161.2021.2013977</pub-id>
          <pub-id pub-id-type="medline">35048782</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="web">
          <article-title>Psychological counseling data</article-title>
          <source>AI Hub</source>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.aihub.or.kr/aihubdata/data/view.do?currMenu=115&#38;dataSetSn=71806&#38;topMenu=100">https://www.aihub.or.kr/aihubdata/data/view.do?currMenu=115&#38;dataSetSn=71806&#38;topMenu=100</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="web">
          <article-title>Data description for psychological counseling data</article-title>
          <source>AI Hub</source>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.aihub.or.kr/aihubdata/data/view.do?currMenu=115&#38;dataSetSn=71806&#38;topMenu=100#:~:text=%EB%8D%B0%EC%9D%B4%ED%84%B0%20%EA%B5%AC%EC%B6%95%20%EA%B7%9C%EB%AA%A8%20%EB%B0%8F%20%EB%8D%B0%EC%9D%B4%ED%84%B0%20%EB%B6%84%ED%8F%AC">https://www.aihub.or.kr/aihubdata/data/view.do?currMenu=115&#38;dataSetSn=71806&#38;topMenu=100#:~:text=%EB%8D%B0%EC%9D%B4%ED%84%B0%20%EA%B5%AC%EC%B6% 95%20%EA%B7%9C%EB%AA%A8%20%EB%B0%8F%20%EB%8D%B0%EC%9D%B4%ED%84%B0%20%EB%B6%84%ED%8F%AC</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="web">
          <article-title>Guideline for use of psychological counseling data</article-title>
          <source>AI Hub</source>
          <year>2024</year>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aihub.or.kr/aihubdata/data/view.do?aihubDataSe=data&#38;currMenu=115&#38;dataSetSn=71806&#38;topMenu=100">https://aihub.or.kr/aihubdata/data/view.do?aihubDataSe=data&#38;currMenu=115&#38;dataSetSn=71806&#38;topMenu=100</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="web">
          <article-title>Classification and generation reference model documentation for psychological counseling data</article-title>
          <source>AI Hub</source>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aihub.or.kr/aihubdata/data/view.do?aihubDataSe=data&#38;currMenu=115&#38;dataSetSn=71806&#38;srchDataRealmCode=REALM006&#38;topMenu=100">https://aihub.or.kr/aihubdata/data/view.do?aihubDataSe=data&#38;currMenu=115&#38;dataSetSn=71806&#38;srchDataRealmCode=REALM006&#38;topMenu=100</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Devlin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>MW</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Toutanova</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Burstein</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Doran</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Solorio</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title>
          <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</source>
          <year>2019</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>4171</fpage>
          <lpage>86</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Park</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Moon</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>WI</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>JY</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Oh</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Oh</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lyu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jeong</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Seo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Jang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Do</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Oh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ha</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>KLUE: Korean Language Understanding Evaluation</article-title>
          <source>Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks</source>
          <year>2021</year>
          <conf-name>NeurIPS 2021</conf-name>
          <conf-date>December 6-14, 2021</conf-date>
          <conf-loc>Virtual Event</conf-loc>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/98dce83da57b0395e163467c9dae521b-Abstract-round2.html"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>World Medical Association</collab>
          </person-group>
          <article-title>World Medical Association Declaration of Helsinki: ethical principles for medical research involving human participants</article-title>
          <source>JAMA</source>
          <year>2025</year>
          <month>01</month>
          <day>07</day>
          <volume>333</volume>
          <issue>1</issue>
          <fpage>71</fpage>
          <lpage>4</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2024.21972</pub-id>
          <pub-id pub-id-type="medline">39425955</pub-id>
          <pub-id pub-id-type="pii">2825290</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="web">
          <article-title>Bioethics and Safety Act</article-title>
          <source>Korea Legislation Research Institute</source>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://elaw.klri.re.kr/eng_mobile/viewer.do?hseq=68583&#38;key=36&#38;type=part">https://elaw.klri.re.kr/eng_mobile/viewer.do?hseq=68583&#38;key=36&#38;type=part</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eysenbach</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Till</surname>
              <given-names>JE</given-names>
            </name>
          </person-group>
          <article-title>Ethical issues in qualitative research on internet communities</article-title>
          <source>BMJ</source>
          <year>2001</year>
          <month>11</month>
          <day>10</day>
          <volume>323</volume>
          <issue>7321</issue>
          <fpage>1103</fpage>
          <lpage>5</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/11701577"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj.323.7321.1103</pub-id>
          <pub-id pub-id-type="medline">11701577</pub-id>
          <pub-id pub-id-type="pmcid">PMC59687</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Middleton</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Sittig</surname>
              <given-names>DF</given-names>
            </name>
            <name name-style="western">
              <surname>Wright</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Clinical decision support: a 25 year retrospective and a 25 year vision</article-title>
          <source>Yearb Med Inform</source>
          <year>2016</year>
          <month>08</month>
          <day>02</day>
          <volume>Suppl 1</volume>
          <issue>Suppl 1</issue>
          <fpage>S103</fpage>
          <lpage>16</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.15265/IYS-2016-s034"/>
          </comment>
          <pub-id pub-id-type="doi">10.15265/IYS-2016-s034</pub-id>
          <pub-id pub-id-type="medline">27488402</pub-id>
          <pub-id pub-id-type="pii">me2016-s034</pub-id>
          <pub-id pub-id-type="pmcid">PMC5171504</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sittig</surname>
              <given-names>DF</given-names>
            </name>
            <name name-style="western">
              <surname>Wright</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Osheroff</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Middleton</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Teich</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Ash</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Campbell</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
          </person-group>
          <article-title>Grand challenges in clinical decision support</article-title>
          <source>J Biomed Inform</source>
          <year>2008</year>
          <month>04</month>
          <volume>41</volume>
          <issue>2</issue>
          <fpage>387</fpage>
          <lpage>92</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(07)00104-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2007.09.003</pub-id>
          <pub-id pub-id-type="medline">18029232</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(07)00104-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC2660274</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Obermeyer</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Powers</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Vogeli</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mullainathan</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title>
          <source>Science</source>
          <year>2019</year>
          <month>10</month>
          <day>25</day>
          <volume>366</volume>
          <issue>6464</issue>
          <fpage>447</fpage>
          <lpage>53</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://escholarship.org/uc/item/6h92v832"/>
          </comment>
          <pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id>
          <pub-id pub-id-type="medline">31649194</pub-id>
          <pub-id pub-id-type="pii">366/6464/447</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="web">
          <article-title>Recommendations for the conduct, reporting, editing, and publication of scholarly work in medical journals</article-title>
          <source>International Committee of Medical Journal Editors</source>
          <year>2026</year>
          <month>01</month>
          <access-date>2026-03-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.icmje.org/recommendations/">https://www.icmje.org/recommendations/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
