<?xml version="1.0" encoding="utf-8"?>
<?xml-model href="https://zfdg.de/sites/default/files/medien/zfdg.rng" type="application/xml" schematypens="http://relaxng.org/ns/structure/1.0"?>
<?xml-model href="https://zfdg.de/sites/default/files/medien/zfdg.rng" type="application/xml" schematypens="http://purl.oclc.org/dsdl/schematron"?>
<TEI xmlns="http://www.tei-c.org/ns/1.0" xmlns:tei="http://www.tei-c.org/ns/1.0">
    <teiHeader>
        <fileDesc>
            <titleStmt>
                <title level="a" type="full">On the Explainability of Vision-Language Models in Art
                    History</title>
                <title level="a" type="short">On the Explainability of Vision-Language Models in Art
                    History</title>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/aut">Author</resp>
                    <resp ref="https://credit.niso.org/contributor-roles/conceptualization/"
                        >Conceptualization</resp>
                    <resp ref="https://credit.niso.org/contributor-roles/investigation/"
                        >Investigation</resp>
                    <resp ref="https://credit.niso.org/contributor-roles/methodology/"
                        >Methodology</resp>
                    <resp ref="https://credit.niso.org/contributor-roles/visualization/"
                        >Visualization</resp>
                    <resp ref="https://credit.niso.org/contributor-roles/writing-original-draft/"
                        >Writing&#160;– original draft</resp>
                    <persName>
                        <forename>Stefanie</forename>
                        <surname>Schneider</surname>
                        <email>stefanie.schneider@uni-marburg.de</email>
                        <idno type="gnd">1220379301</idno>
                        <idno type="orcid">0000-0003-4915-6949</idno>
                        <affiliation>Phillips-Universität Marburg</affiliation>
                    </persName>
                </respStmt>
            </titleStmt>
            <editionStmt>
                <edition n="1.0"/>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/dtm">Technische
                        Redaktion</resp>
                    <persName>
                        <forename>Martin</forename>
                        <surname>de la Iglesia</surname>
                        <idno type="gnd">1095143719</idno>
                        <idno type="orcid">0000-0002-9319-4793</idno>
                    </persName>
                </respStmt>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/dtm">Technische
                        Redaktion</resp>
                    <persName>
                        <forename>Maximilian</forename>
                        <surname>Görmar</surname>
                        <idno type="gnd">1077317964</idno>
                        <idno type="orcid">0000-0003-3608-1140</idno>
                    </persName>
                </respStmt>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/pfr">Textredaktion</resp>
                    <persName>
                        <forename>Karoline</forename>
                        <surname>Lemke</surname>
                        <idno type="gnd">1187840033</idno>
                        <idno type="orcid">0000-0002-1604-672X</idno>
                    </persName>
                </respStmt>
            </editionStmt>
            <publicationStmt>
                <publisher n="Redaktionssitz">
                    <orgName>Herzog August Bibliothek</orgName>
                    <address>
                        <addrLine>Lessingplatz 1</addrLine>
                        <addrLine>38304 Wolfenbüttel</addrLine>
                    </address>
                </publisher>
                <publisher n="herausgebendes Organ">
                    <orgName>Forschungsverbund Marbach Weimar Wolfenbüttel</orgName>
                    <address>
                        <addrLine>Burgplatz 4</addrLine>
                        <addrLine>99423 Weimar</addrLine>
                    </address>
                </publisher>
                <publisher n="herausgebendes Organ">
                    <orgName>Digital Humanities im deutschsprachigen Raum e. V.</orgName>
                    <address>
                        <addrLine>Hamburg</addrLine>
                    </address>
                </publisher>
                <date n="1.0" when="2026-09-30">30.09.2026</date>
                <idno type="doi">10.17175/sb009_002</idno>
                <idno type="ppn">1982298022</idno>
                <availability status="free">
                    <licence target="https://creativecommons.org/licenses/by-sa/4.0/">CC BY-SA 4.0,
                        sofern nicht anders angegeben.</licence>
                </availability>
            </publicationStmt>
            <seriesStmt>
                <title level="j">Zeitschrift für digitale Geisteswissenschaften</title>
                <title level="m">(Generative) KI für Kultur- und Textdaten</title>
                <title level="s">Sonderbände</title>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/edt">Editor</resp>
                    <persName>
                        <forename>Gerrit</forename>
                        <surname>Brüning</surname>
                        <email>gerrit.bruening@klassik-stiftung.de</email>
                        <idno type="gnd">1070804975</idno>
                        <idno type="orcid">0000-0003-2402-6734</idno>
                        <affiliation>Klassik Stiftung Weimar, Goethe- und Schiller-Archiv, Abteilung
                            Digitale Editionen</affiliation>
                    </persName>
                </respStmt>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/edt">Editor</resp>
                    <persName>
                        <forename>Sarah</forename>
                        <surname>Oberbichler</surname>
                        <email>Sarah.oberbichler@uni.lu</email>
                        <idno type="gnd">122356195X</idno>
                        <idno type="orcid">0000-0002-1031-2759</idno>
                        <affiliation>Luxembourg Centre for Contemporary and Digital History
                            (C²DH)</affiliation>
                    </persName>
                </respStmt>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/edt">Editor</resp>
                    <persName>
                        <forename>Cindarella</forename>
                        <surname>Petz</surname>
                        <email>petz@ieg-mainz.de</email>
                        <idno type="gnd">1245584375</idno>
                        <idno type="orcid">0000-0002-6178-7332</idno>
                        <affiliation>DH Lab, Leibniz-Institut für Europäische Geschichte
                            Mainz</affiliation>
                    </persName>
                </respStmt>
                <respStmt>
                    <resp ref="http://id.loc.gov/vocabulary/relators/edt">Editor</resp>
                    <persName>
                        <forename>Pia</forename>
                        <surname>Schwarz</surname>
                        <email>schwarz@ids-mannheim.de</email>
                        <idno type="gnd">1392456525</idno>
                        <idno type="orcid">0000-0002-3339-7476</idno>
                        <affiliation>Leibniz-Institut für Deutsche Sprache</affiliation>
                    </persName>
                </respStmt>
                <idno type="issn">2510-1366</idno>
                <idno type="ppn">1982260998</idno>
                <idno type="doi">10.17175/sb009</idno>
                <idno type="url">https://www.zfdg.de/sonderband/9</idno>
                <biblScope unit="specialvolume">9</biblScope>
                <biblScope unit="article">2</biblScope>
            </seriesStmt>
            <sourceDesc>
                <p>Born digital: no previous source exists.</p>
            </sourceDesc>
        </fileDesc>
        <encodingDesc>
            <editorialDecl>
                <p>Letzte Überprüfung aller Verweise: <date when="2026-08-27">27.08.2026</date>
                </p>
            </editorialDecl>
            <schemaRef url="https://zfdg.de/sites/default/files/medien/zfdg.odd"/>
        </encodingDesc>
        <profileDesc>
            <textClass>
                <keywords n="Beitragstyp">
                    <term>Fachartikel</term>
                </keywords>
                <keywords n="GND">
                    <term ref="https://d-nb.info/gnd/4145391-8">Bildanalyse</term>
                    <term ref="https://d-nb.info/gnd/1263068472">Erklärbare künstliche Intelligenz</term>
                    <term ref="https://d-nb.info/gnd/4138803-3">Kunstgeschichte</term>                   
                </keywords>
            </textClass>
        </profileDesc>
    </teiHeader>
    <text xml:lang="en">
        <front>
            <div type="abstract" xml:lang="en">
                <p><term type="dh">Vision-Language Models (VLMs)</term> transfer visual and textual
                    data into a shared embedding space. In doing so, they enable a wide range of
                    multimodal tasks, while also raising critical questions about the nature of
                    machine ›understanding‹. In this paper, we examine how <term type="dh"
                        >Explainable Artificial Intelligence (XAI)</term> methods can render the
                    visual reasoning of a VLM&#160;– namely, CLIP&#160;– legible in art-historical
                    contexts. To this end, we evaluate seven methods, combining zero-shot
                    localization experiments with human interpretability studies. Our results
                    indicate that, while these methods capture some aspects of human interpretation,
                    their effectiveness hinges on the conceptual stability and representational
                    availability of the examined categories.</p>
            </div>
            <div type="abstract" xml:lang="de">
                <p><term type="dh">Vision-Language Models (VLMs)</term> überführen visuelle und
                    textuelle Daten in einen gemeinsamen Einbettungsraum. Dabei ermöglichen sie eine
                    Vielzahl multimodaler Aufgaben, werfen jedoch zugleich kritische Fragen nach der
                    Natur des maschinellen ›Verstehens‹ auf. In diesem Beitrag untersuchen wir, wie
                    Methoden der <term type="dh">Explainable Artificial Intelligence (XAI)</term>
                    das visuelle Schlussfolgern eines VLM&#160;– konkret CLIP&#160;– in
                    kunsthistorischen Kontexten nachvollziehbar machen können. Zu diesem Zweck
                    evaluieren wir sieben Methoden, die Zero-Shot-Lokalisierungsexperimente mit
                    Studien zur menschlichen Interpretierbarkeit kombinieren. Unsere Ergebnisse
                    zeigen, dass diese Methoden zwar einige Aspekte menschlicher Interpretation
                    erfassen, ihre Wirksamkeit jedoch von der konzeptuellen Stabilität und der
                    repräsentationalen Verfügbarkeit der untersuchten Kategorien abhängt.</p>
            </div>
        </front>
        <body>
            <div type="chapter">
                <head>1. Introduction</head>
                <p>In recent years, <term type="dh">Vision-Language Models (VLMs)</term> have become
                    remarkably versatile instruments of analysis. By aligning visual and linguistic
                    information within a shared embedding space, they can perform a wide range of
                    multimodal tasks&#160;– from retrieval<note type="footnote">Cf. <ref
                            type="bibliography" target="#radford_et_al_models_2021">Radford
                            et&#160;al. 2021</ref>.</note> and captioning<note type="footnote">Cf.
                            <ref type="bibliography" target="#li_et_al_blip_2023">Li et&#160;al.
                            2023</ref>.</note> to zero-shot classification.<note type="footnote">Cf.
                            <ref type="bibliography" target="#zhai_et_al_transfer_2022">Zhai
                            et&#160;al. 2022</ref>.</note> However, this versatility has made them
                    the object of sustained criticism: Not only due to the opacity of their internal
                    mechanisms, but also because of the ethical, sociotechnical, and epistemological
                    assumptions encoded in their design. It has been questioned what forms of
                    ›understanding‹ such models enact, how their embeddings reify social
                    hierarchies, and to which extent their apparent generality conceals dependencies
                    on biased, uncurated data.<note type="footnote">Cf. <ref type="bibliography"
                            target="#bhalla_et_al_clip_2024">Bhalla et&#160;al. 2024</ref>; <ref
                            type="bibliography" target="#birhane_et_al_datasets_2021">Birhane
                            et&#160;al. 2021</ref>; <ref type="bibliography"
                            target="#bender_et_al_language_2021">Bender et&#160;al.
                        2021</ref>.</note> In short, we might ask: What does it mean for a model to
                        <hi rend="italic">see</hi>?</p>
                <p>This question is particularly relevant in fields where visual meaning is
                    historically and semantically dense&#160;– where ›objects‹, in the broadest
                    sense, cannot be reduced to mere labels or descriptive tokens. Art history is
                    exemplary in this regard: Here, the visual is not simply perceived, but
                    interpreted through culturally sedimented conventions of style, iconography, and
                    material practice. Nevertheless, models such as <term type="dh">CLIP
                        (Contrastive Language–Image Pre-training)</term>
                    <note type="footnote">Cf. <ref type="bibliography"
                            target="#radford_et_al_models_2021">Radford et&#160;al.
                        2021</ref>.</note> are now routinely employed ›out of the box‹ for
                    art-historical retrieval and analysis on digital platforms,<note type="footnote"
                        >Cf. <ref type="bibliography" target="#springstein_et_al_iART_2021"
                            >Springstein et&#160;al. 2021</ref>; <ref type="bibliography"
                            target="#offert_bell_imgs_2023">Offert&#160;/ Bell 2023</ref>.</note>
                    often without a clear understanding of which kinds of visual concepts&#160;–
                    formal, iconographic, or affective&#160;– are encoded in their embeddings.</p>
                <p>CLIP is a VLM that aligns images and texts within a shared embedding space.
                    Trained on a large set of image-text pairs scraped from the web, it learns to
                    associate visual and linguistic patterns by grouping similar representations and
                    separating dissimilar ones. Large-scale web-scraped datasets such as
                        LAION-400M<note type="footnote">Cf. <ref type="bibliography"
                            target="#schuhmann_et_al_laion_2021">Schuhmann et&#160;al.
                        2021</ref>.</note>, however, are by no means neutral repositories of visual
                    culture. As Abeba Birhane, Vinay Uday Prabhu, and Emmanuel Kahembwe demonstrate,
                    such datasets contain structural biases, non-consensual imagery, and
                    stereotypical or pornographic representations that reflect the discriminatory
                    nature of the web.<note type="footnote">Cf. <ref type="bibliography"
                            target="#birhane_et_al_datasets_2021">Birhane et&#160;al.
                        2021</ref>.</note> The epistemic opacity of VLMs thus becomes a
                    methodological issue: How can search results be interpreted when the model
                    itself embeds an unacknowledged theory of vision? CLIP, in particular,
                    epitomizes this multimodal turn: Its embedding space constitutes not only a
                    technical geometry of similarity but what Leonardo Impett and Fabian Offert call
                    a »vector imaginary«<note type="footnote"><ref type="bibliography" target="#impett_offert_arthistory_2022">Impett &#160;/ Offert 2022</ref>.</note> of
                    contemporary visual culture&#160;– a statistical condensation of what is
                    collectively pictured and named online. This makes CLIP both uniquely powerful
                    and uniquely problematic for art-historical inquiry. On the one hand, its
                    zero-shot capacity enables the retrieval of artworks that are stylistically or
                    iconographically related without supervision&#160;– in other words, it can
                    identify similarities even among categories that it was never explicitly trained
                    on. Yet the same mechanism also perpetuates the omissions of its training
                    corpus, reproducing a visuality that is historically and culturally uneven. </p>
                <p>Against this backdrop, we ask a central question: To what extent can <term
                        type="dh">Explainable Artificial Intelligence (XAI)</term> methods render
                    the visual logic of CLIP legible to human interpreters, thereby strengthening
                    the methodological robustness of VLMs in art-historical contexts? XAI refers to
                    a variety of techniques designed to elucidate model behavior, including post hoc
                    attribution, concept-based analysis, and related methods that indicate which
                    features of an input contribute most strongly to a given output.<note
                        type="footnote"> See, e.g., <ref type="bibliography"
                            target="#speith_review_2022">Speith 2022</ref>; <ref type="bibliography"
                                target="#saeed_omlin_xai_2023">Saeed&#160;/ Omlin 2023</ref>; <ref type="bibliography" target="#hassija_et_al_black-box_2024">Hassija et&#160;al. 2024.</ref></note> To explore this question, we comparatively evaluate seven
                    XAI methods spanning three paradigms: (1) Gradient-based methods that
                    backpropagate class-specific gradients into feature maps (Grad-CAM, Grad-CAM++,
                    LayerCAM, and LeGrad);<note type="footnote">Cf. <ref type="bibliography"
                            target="#selvaraju_et_al_grad-cam_2017">Selvaraju et&#160;al.
                        2017</ref>; <ref type="bibliography"
                            target="#chattopadhyay_et_al_gradCAM_2018">Chattopadhyay et&#160;al.
                            2018</ref>; <ref type="bibliography" target="#jiang_et_al_maps_2021"
                            >Jiang et&#160;al. 2021</ref>; <ref type="bibliography"
                            target="#bousselham_et_al_explainability_2024">Bousselham et&#160;al.
                            2024</ref>. CAM denotes Class Activation Mapping.</note> (2) score-based, gradient-free methods that measure
                    the influence of image regions on the model-predicted score (ScoreCAM and
                        gScoreCAM);<note type="footnote">Cf. <ref type="bibliography"
                            target="#wang_et_al_score-CAM_2020">Wang et&#160;al. 2020</ref>; <ref
                            type="bibliography" target="#chen_et_al_gScoreCAM_2022">Chen et&#160;al.
                            2022</ref>.</note> and (3) CLIP-specific approaches that intervene
                    directly in the inference pipeline (CLIP Surgery).<note type="footnote">Cf. <ref
                            target="#li_et_al_explainability_2025">Li et&#160;al. 2025</ref>.</note>
                    Each of these methods generates a saliency map for a given text prompt that
                    visualizes the contribution of an image region to the model’s predicted score
                        (<ref type="graphic" target="#vlm_ArtHistory_001">Fig.&#160;1</ref>). In
                    this paper, our objective is not just to identify the most effective technique
                    but also to explore the boundaries of explainability under conditions of
                    zero-shot inference and domain transfer. Thus, we included methods that, while
                    no longer state-of-the-art in some cases, remain prevalent in practice, in order
                    to highlight the interpretive and subjective dimensions of ›explainability‹
                    itself.</p>
                <figure>
                    <graphic xml:id="vlm_ArtHistory_001" url="Medien/vlm_ArtHistory_001.png">
                        <desc>
                            <ref type="intern" target="#abb1">Figure&#160;1</ref>: The saliency map
                            highlights, in red, the image regions most strongly associated with the
                            concept of the ›snake‹ in Franz von Stuck’s <title>Adam and Eve</title>
                            (c. 1920). [Visualization: Stefanie Schneider 2026]</desc>
                    </graphic>
                </figure>
                <p>To examine these dimensions systematically, we adopted a two-stage evaluation
                    framework. First, a quantitative case study employed two art-historical datasets
                    &#160;– IconArt<note type="footnote">Cf. <ref type="bibliography"
                            target="#gonthier_et_al_computervision_2018">Gonthier et&#160;al.
                            2018</ref>.</note> and ArtDL<note type="footnote">Cf. <ref
                            type="bibliography" target="#milani_fraternali_dataset_2021"
                            >Milani&#160;/ Fraternali 2021</ref>.</note>&#160;– to measure the
                    localization accuracy of the methods under zero-shot conditions. Then, in an
                    online survey with participants trained in art history, the interpretability of
                    these same methods was evaluated, situating the previously obtained results
                    within the variability of human visual judgment. Building on the work of Usha
                    Bhalla, Alex Oesterling, Suraj Srinivas, Flávio P. Calmon and Himabindu
                    Lakkaraju, who demonstrate that CLIP embeddings can be decomposed into sparse,
                    interpretable concept spaces,<note type="footnote">Cf. <ref type="bibliography"
                            target="#bhalla_et_al_clip_2024">Bhalla et&#160;al. 2024</ref>.</note>
                    our studies asked whether such interpretability extends to visual explanations:
                    that is, do XAI methods disclose a model’s internal conceptual structure&#160;–
                    or merely <hi rend="italic">aestheticize</hi> its opacity? This question is
                    especially pertinent given critiques by Birhane et al. and Emily M. Bender and
                    her Co-Authors Timnit Gebru, Angelina McMillan-Major, and Shmargaret Shmitchell,
                    who argue that large-scale, web-scraped datasets can perpetuate hegemonic biases
                    under the guise of generality.<note type="footnote">Cf. <ref type="bibliography"
                            target="#birhane_et_al_datasets_2021">Birhane et&#160;al. 2021</ref>;
                            <ref type="bibliography" target="#bender_et_al_language_2021">Bender
                            et&#160;al. 2021</ref>.</note> If explainability techniques cannot
                    illuminate these latent structures, they may reiterate rather than expose the
                    ideological patterns of machine vision.</p>
                <p>From this standpoint, we articulate three research questions: (1) How effectively
                    do XAI methods localize iconographic objects in artworks under zero-shot
                    conditions without fine-tuning? This establishes a baseline: Can methods trained
                    on everyday imagery nonetheless delineate the complex, symbolically charged
                    forms within artworks beyond their training distribution? (2) Does the visual
                    relevance of these maps correspond to human judgments? If saliency maps claim to
                    visualize ›what the model sees‹, their validity must be tested against human
                    perception&#160;– specifically against the art-historically informed gaze. (3)
                    Which factors, such as object size and concept abstraction, drive performance
                    differences? Here, we connected measurable attributes with semantic ones,
                    aligning computational error analysis with questions of representation central
                    to art-historical inquiry. By testing how&#160;– and where&#160;– XAI methods
                    succeed or fail in making CLIP’s mechanisms visible, we contributed to a broader
                    methodological debate about how digital art history might critically engage with
                    the epistemic structures of machine vision. Our aim, in other words, is to
                    determine not only what VLMs attend to in works of art but examine how their
                    patterns of attention either align&#160;– or fail to align&#160;– with human
                    interpretive conventions.</p>
                <p>The remainder of this paper is structured as follows: In <ref type="intern"
                        target="#hd2">Section 2</ref>, we outline the rationale for selecting the
                    examined XAI techniques. <ref type="intern" target="#hd3">Section 3</ref>
                    presents the first case study, which quantitatively evaluates localization
                    accuracy using two art-historical datasets under zero-shot conditions. <ref
                        type="intern" target="#hd7">Section 4</ref> turns to the second case study:
                    an online survey that assesses the interpretability of saliency maps from an
                    art-historical perspective. <ref type="intern" target="#hd11">Section 5</ref>
                    then synthesizes these findings, tracing their methodological implications for
                    explainability, interpretability, and the critical analysis of machine vision
                    within digital art history.</p>
            </div>
            <div type="chapter">
                <head>2. Method Selection</head>
                <p>Since the advent of <term type="dh">Artificial Intelligence (AI)</term>,
                    researchers have sought to make its internal mechanisms intelligible. This
                    imperative has only intensified further with the resurgence of <term type="dh"
                        >Deep Neural Networks (DNNs)</term> in the 2000s&#160;– highly non-linear
                    statistical models with millions of parameters. The demand for explanation,
                    however, predates deep learning; it can be traced back to the early attempts to
                    understand expert systems,<note type="footnote">E.g., <ref type="bibliography"
                            target="#buchanan_shortliffe_eds_expertsystems_1985">Buchanan&#160;/
                            Shortliffe (eds.) 1985</ref>.</note> which, emerging in the 1960s and
                    1970s, marked one of the first conceptual turns in AI towards the explainability
                    of machine reasoning.<note type="footnote">Cf. <ref type="bibliography"
                            target="#coy_bonsiepen_expertensystemtechnik_1989">Coy&#160;/ Bonsiepen
                            1989</ref>.</note> Designed to mimic the problem-solving strategies of
                    human specialists within narrowly defined use cases, these systems aimed not
                    only to provide decisions, but also to explain the reasoning behind them.<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#harmon_et_al_expertensysteme_1989">Harmon et&#160;al.
                            1989</ref>.</note> Their architecture&#160;&#160;– typically comprising
                    a knowledge base and an inference engine&#160;– explicitly separated
                    domain-specific expertise from domain-independent reasoning procedures.<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#puppe_expertensystem_1988">Puppe 1988</ref>.</note> Although
                    modern DNNs have largely abandoned such symbolic, rule-based representations in
                    favor of statistical learning, the epistemological tension first articulated in
                    the age of expert systems&#160;– between algorithmic performance and
                    interpretability&#160;– remains a defining problem for contemporary research in
                    XAI.</p>
                <p>In this section, however, we do not intend to provide a historical or taxonomic
                    overview of XAI methods&#160;– an exercise already undertaken in recent years
                    from a variety of perspectives.<note type="footnote">E.g., <ref
                            type="bibliography" target="#speith_review_2022">Speith 2022</ref>; <ref
                            type="bibliography" target="#saeed_omlin_xai_2023">Saeed&#160;/ Omlin
                            2023</ref>; <ref type="bibliography"
                            target="#hassija_et_al_black-box_2024">Hassija et&#160;al.
                        2024</ref>.</note> Rather, we motivate our decision to focus on a specific
                    class of <hi rend="italic">perceptive interpretability</hi> methods, i.e.,
                    saliency-based visualization methods applied post hoc to a pre-trained model, in
                    our case CLIP. Our goal is not to revisit the broader epistemological debates
                    surrounding explainability, but to examine how visual explanations&#160;– those
                    that make a model’s internal reasoning visible&#160;– operate in the specific
                    and practically relevant context of VLMs. In this context, perceptive
                    interpretability methods, and saliency-based visualizations in particular,
                    occupy a unique position at the interface of human perception and machine
                    inference, as they generate intuitive, human-readable representations of model
                    attention. In doing so, they provide a spatial grammar through which one might
                    then ask where a model ›looks‹ when associating textual prompts with visual
                    content. We therefore selected methods according to three criteria: (1) The
                    method must produce spatially localized, human-inspectable heatmaps over the
                    input image; (2) it must operate post hoc on CLIP&#160;– without retraining or
                    architectural modification&#160;– to ensure comparability across prompts and
                    datasets; and (3) it must be either widely adopted as a baseline or specifically
                    adapted to CLIP’s dual-encoder architecture, making text–image interactions
                    explicit. We excluded approaches that introduce additional hyperparameters, such
                    as CLIP-LIME<note type="footnote">Cf. <ref type="bibliography"
                            target="#kazmierczak_et_al_clip_2023">Kazmierczak et&#160;al.
                        2023</ref>.</note>. Likewise, attention-based methods<note type="footnote"
                        >E.g., <ref type="bibliography" target="#abnar_zidema_attention_2020"
                            >Abnar&#160;/ Zuidema 2020</ref>.</note> were omitted, as their reliance
                    on attention weights has been shown to correlate only weakly with decision
                        relevance.<note type="footnote">Cf. <ref type="bibliography"
                            target="#jain_wallace_attention_2019">Jain&#160;/ Wallace
                        2019</ref>.</note> Also excluded were methods such as CLIP-Dissect<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#oikarinen_weng_clip_2023">Oikarinen&#160;/ Weng
                        2023</ref>.</note>: Although they yield valuable insights into the
                    representational topology of CLIP, they do not address the specific perceptual
                    question that motivates our study, i.e., where in the image does the model
                    locate the evidence for a given prompt?</p>
                <p>Given these constraints, we have identified three methodological paradigms for
                    evaluation, each of which renders the most relevant image regions visible to a
                    model in a different way: <hi rend="italic">Gradient-based</hi> methods trace
                    how the model’s internal signals&#160;– its gradients&#160;– shift in response
                    to visual stimuli, revealing which regions of an image most strongly determine a
                    given decision; <hi rend="italic">score-based</hi>, gradient-free methods, by
                    contrast, perturb the input: They mask regions of the image and measure the
                    resulting change in the model’s output, inferring from these changes which areas
                    carry the greatest weight; CLIP-specific interventions operate directly on the
                    multimodal architecture, adjusting how CLIP integrates textual and visual
                    information to make visible the relational mechanics between words and image
                    regions. Within the gradient-based family, we include Grad-CAM<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#selvaraju_et_al_grad-cam_2017">Selvaraju et&#160;al.
                        2017</ref>.</note> as the canonical baseline, Grad-CAM++<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#chattopadhyay_et_al_gradCAM_2018">Chattopadhyay et&#160;al.
                            2018</ref>.</note> as its extension to multi-instance scenarios via
                    higher-order gradient weighting, and LayerCAM<note type="footnote">Cf. <ref
                            type="bibliography" target="#jiang_et_al_maps_2021">Jiang et&#160;al.
                            2021</ref>.</note> as a refinement that pools activations from
                    intermediate convolutional layers to enhance spatial fidelity. LeGrad<note
                        type="footnote">Cf. <ref type="bibliography"
                            target="#bousselham_et_al_explainability_2024">Bousselham et&#160;al.
                            2024</ref>.</note> further optimizes this aggregation process, improving
                    localization accuracy while reducing sensitivity to layer selection. The second
                    paradigm&#160;– score-based, gradient-free evaluation&#160;– is exemplified by
                        ScoreCAM<note type="footnote">Cf. <ref type="bibliography"
                            target="#wang_et_al_score-CAM_2020">Wang et&#160;al. 2020</ref>.</note>
                    and gScoreCAM.<note type="footnote">Cf. <ref type="bibliography"
                            target="#chen_et_al_gScoreCAM_2022">Chen et&#160;al. 2022</ref>.</note>
                    Both rely on selective occlusion: By systematically masking portions of the
                    image and recording how the model’s scores change, they construct saliency maps
                    that recompose these perturbations into a spatial account of the model’s visual
                    attention. Finally, CLIP Surgery<note type="footnote">Cf. <ref
                            type="bibliography" target="#li_et_al_explainability_2025">Li
                            et&#160;al. 2025</ref>.</note> represents a model-specific intervention
                    at inference time. By reparameterizing the forward pass and adapting
                    Grad-CAM-like mechanisms to CLIP’s dual-encoder architecture, it produces
                    activation maps that more explicitly disentangle the textual and visual streams
                    of the model&#160;– rendering visible, in effect, how the network’s multimodal
                    alignments are spatially instantiated.</p>
                <p>However, while such techniques have become&#160;– at least to some extent&#160;–
                    standard practice within machine vision, their epistemic foundations remain
                    contested. According to Tim Miller, explanations are not static products but
                    dialogical processes that unfold within particular social contexts: They
                    presuppose both an <hi rend="italic">explainer</hi> and an <hi rend="italic"
                        >explainee</hi>, whose beliefs, expectations, and cognitive biases shape not
                    only the form of an explanation, but also how adequate it is perceived to
                        be.<note type="footnote">Cf. <ref type="bibliography"
                            target="#miller_explanation_2019">Miller 2019</ref>.</note> As Miller
                    emphasizes, explanations are inherently contrastive (<hi rend="italic">Why this
                        rather than that?</hi>), selective (<hi rend="italic">Which of the many
                        possible causes should be prioritized?</hi>), and social (<hi rend="italic"
                        >How should this explanation be shaped for the person to whom it is
                        addressed?</hi>). In this sense, Miller’s ›social model of explanation‹
                    provides a conceptual bridge between the epistemic pragmatics of XAI and the
                    interpretive negotiations that characterize humanistic inquiry. For
                    art-historical applications&#160;– where meaning is constructed through dialogue
                    rather than extracted from data&#160;– this social-cognitive model is especially
                    pertinent. We, thus, combined quantitative and qualitative experiments to
                    examine how explanation methods function not only as diagnostic instruments but
                    as interfaces between human vision and machine inference.</p>
            </div>
            <div type="chapter">
                <head>3. Case Study 1</head>
                <p>In the first case study, we evaluated the zero-shot localization capabilities of
                    CLIP by examining the afore-introduced XAI techniques using large-scale datasets
                    comprising nearly 2,000 art-historical images in total. In so doing, we
                    determined how effectively each method can identify and delineate objects in
                    domain-specific imagery without any fine-tuning. The aim of this study was to
                    establish a quantitative foundation for assessing how different explainability
                    methods perform in visually and semantically complex cultural datasets, and to
                    identify which approaches transfer most effectively to art-historical imagery.
                    Our analysis focused on comparing these techniques&#160;– Grad-CAM, Grad-CAM++,
                    LayerCAM, LeGrad, ScoreCAM, gScoreCAM, and CLIP Surgery&#160;– under consistent
                    experimental conditions, thereby delineating the empirical contours against
                    which subsequent interpretive analyses (<ref type="intern" target="#hd7">Section
                        4</ref>) can be situated.</p>
                <div type="subchapter">
                    <head>3.1 Pre-processing Steps</head>
                    <p>For each method, we generated a class-conditional saliency map, thresholded
                        it at τ to obtain a binary mask, and extracted the tightest box around the
                        largest connected component (<ref type="graphic"
                            target="#vlm_ArtHistory_002">Fig.&#160;2</ref>). These boxes were then
                        compared to ground-truth annotations from the datasets described in <ref
                            target="#hd5">Section 3.2.</ref> While fixed thresholds (e.g., τ = 0.2)
                        can yield plausible maps, they are notoriously sensitive&#160;– to the
                        explainability method, the dataset, and even the target class
                        semantics&#160;– leading to unstable conclusions. Following Junsuk Choe,
                        Seong Joon Oh, Seungho Lee, Sanghyuk Chun, Zeynep Akata and Hyunjung
                            Shim,<note type="footnote">Cf. <ref type="bibliography"
                                target="#choe_et_al_methods_2020">Choe et al. 2020</ref>.</note> we
                        therefore adopted a threshold-independent evaluation: For each method, we
                        computed the maximum BoxAcc, which denotes bounding-box localization accuracy at a specified <term type="dh">Intersection-over-Union (IoU)</term> threshold, over τ ∈ [0.2, 0.9] via grid search, ensuring
                        that it was evaluated under the most favorable operating conditions. Unlike
                        Choe et al., who average BoxAcc across IoU thresholds δ ∈ {0.3, 0.5, 0.7},<note type="footnote"
                            >Cf. <ref type="bibliography" target="#choe_et_al_methods_2020">Choe et
                                al. 2020</ref>.</note> we reported BoxAcc separately at each δ to
                        identify more granular differences in localization quality. Formally:
                            <lb/><formula notation="mathml">
                            <math xmlns="http://www.w3.org/1998/Math/MathML" display="inline">
                                <mrow>
                                    <mi mathvariant="italic">BoxAcc</mi>
                                    <mo stretchy="false">(</mo>
                                    <mi>τ</mi>
                                    <mi>,</mi>
                                    <mi>δ</mi>
                                    <mo stretchy="false">)</mo>
                                    <mtext>=</mtext>
                                    <mfrac>
                                        <mn>1</mn>
                                        <mi>N</mi>
                                    </mfrac>
                                    <mrow>
                                        <munder>
                                            <mo stretchy="false">∑</mo>
                                            <mi>n</mi>
                                        </munder>
                                    </mrow>
                                    <msub>
                                        <mn>1</mn>
                                        <mrow>
                                            <mi mathvariant="italic">IoU</mi>
                                            <mo stretchy="false">(</mo>
                                            <mi mathvariant="italic">box</mi>
                                            <mo stretchy="false">(</mo>
                                            <mi>s</mi>
                                            <mo stretchy="false">(</mo>
                                            <msup>
                                                <mi>X</mi>
                                                <mrow>
                                                  <mo stretchy="false">(</mo>
                                                  <mi>n</mi>
                                                  <mo stretchy="false">)</mo>
                                                </mrow>
                                            </msup>
                                            <mo stretchy="false">)</mo>
                                            <mi>,</mi>
                                            <mi>τ</mi>
                                            <mo stretchy="false">)</mo>
                                            <mi>,</mi>
                                            <msup>
                                                <mi>B</mi>
                                                <mrow>
                                                  <mo stretchy="false">(</mo>
                                                  <mi>n</mi>
                                                  <mo stretchy="false">)</mo>
                                                </mrow>
                                            </msup>
                                            <mo stretchy="false">)</mo>
                                            <mi>≥</mi>
                                            <mi>δ</mi>
                                        </mrow>
                                    </msub>
                                </mrow>
                            </math>
                        </formula>
                        <lb/>where <hi rend="italic">s(X<hi rend="super">(n)</hi>)</hi> is the
                        saliency map for image <hi rend="italic">X<hi rend="super">(n)</hi></hi>,
                            <hi rend="italic">box(s,τ)</hi> is the tightest bounding box enclosing
                        the largest-area connected component obtained by thresholding s at τ, and
                            <hi rend="italic">B<hi rend="super">(n)</hi></hi> is the corresponding
                        ground-truth bounding box. In simple terms, BoxAcc indicates whether the
                        highlighted region overlaps sufficiently with the true object location.</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_002" url="Medien/vlm_ArtHistory_002.png">
                            <desc>
                                <ref type="intern" target="#abb2">Figure&#160;2</ref>: For Ercole
                                de’ Roberti’s <title>The Wife of Hasdrubal and Her Children</title>
                                (c. 1490&#160;/ 1493), the ground-truth bounding boxes for the
                                concept ›nudity‹ are shown in green (a). The bounding boxes derived
                                from the class-conditional saliency map are displayed at
                                progressively increasing threshold levels τ ∈ {0.20, 0.30, 0.40,
                                0.50, 0.60}. [Visualization: Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <p>All seven methods were evaluated using compatible CLIP variants: Grad-CAM,
                        Grad-CAM++, LayerCAM, ScoreCAM, gScoreCAM, and CLIP Surgery were applied to
                        the ResNet-50×16 backbone; LeGrad was evaluated on the ViT-B/32 backbone.
                        For the gradient-based methods, features were extracted from the third ReLU
                        activation in the final bottleneck block of the ResNet layer4, or from the
                        last self-attention heads for <term type="dh">Vision Transformer
                            (ViT)</term> models. Gradients of the class score with respect to these
                        activations yielded the channel importance weights. For gradient-free
                        methods, we masked each channel individually by setting all others in the
                        upsampled activation map to zero and computed the cosine similarity between
                        the resulting image embedding and the text prompt to derive the channel
                        importance. CLIP Surgery applies two inference-time modifications&#160;– an
                        adjusted self-attention block and a dual-path feed-forward network&#160;– to
                        mitigate noisy visualizations without requiring any fine-tuning.<note
                            type="footnote">Cf. <ref type="bibliography"
                                target="#li_et_al_explainability_2025">Li et&#160;al.
                            2025</ref>.</note> All code was implemented in PyTorch and ran on two
                        NVIDIA GeForce RTX 2080 Ti.</p>
                </div>
                <div type="subchapter">
                    <head>3.2 Data</head>
                    <p>As Stefanie Schneider and Ricarda Vollmer have observed, object-level
                        annotations&#160;– especially those specifying spatial localization through
                        bounding boxes&#160;– are uncommon in art-historical datasets.<note
                            type="footnote">Cf. <ref type="bibliography"
                                target="#schneider_vollmer_poses_2024">Schneider&#160;/ Vollmer
                                2024</ref>.</note> When such annotations do exist, they often
                        deliberately avoid classes representing iconographic concepts, precisely
                        those motifs whose semantic variability resists stable formalization. The
                        DEArt dataset, for instance, is restricted to broad descriptive categories
                        such as ›people‹ or to visually unambiguous but iconographically neutral
                        objects like ›boats‹ or ›vases‹.<note type="footnote">Cf. <ref
                                type="bibliography" target="#reshetnikov_et_al_deart_2022"
                                >Reshetnikov et&#160;al. 2022</ref>.</note> Yet precisely
                        these unstable iconographic categories most urgently require closer
                        examination if we are to assess whether VLMs can move beyond generic
                        recognition to something resembling art-historical expertise.</p>
                    <p>For this reason, our evaluation focused on two datasets that explicitly
                        provide iconographic content: IconArt<note type="footnote">Cf. <ref
                                type="bibliography" target="#gonthier_et_al_computervision_2018"
                                >Gonthier et&#160;al. 2018</ref>.</note> and ArtDL<note
                            type="footnote">Cf. <ref type="bibliography"
                                target="#milani_fraternali_dataset_2021">Milani&#160;/ Fraternali
                                2021</ref>.</note> (<ref type="graphic" target="#tab_001">Table
                            1</ref>; <ref type="graphic" target="#vlm_ArtHistory_003"
                            >Fig.&#160;3</ref>). Both provide annotations not only for figures
                        (e.g., ›Saint Sebastian‹) but also for attributes and symbols (e.g.,
                        ›Ointment Jar‹). Of these, ArtDL is broader in scope, comprising 10 saints
                        and 49 attributes. However, the usefulness of these datasets as benchmarks
                        is limited by their distribution: Both have pronounced long-tail
                        distributions, with a few frequent classes and many sparsely represented
                        ones. In IconArt, three generic categories&#160;– ›beard‹ (21.98&#160;%),
                        ›angel‹ (21.15&#160;%), and ›nudity‹ (15.39&#160;%)&#160;– account for
                        nearly two-thirds of all annotations, while classes of greater
                        art-historical specificity such as ›Saint Sebastian‹ (1.66&#160;%) or
                        ›Crucifixion of Jesus‹ (2.21&#160;%) appear only rarely. ArtDL displays the
                        same dynamic even more strongly: The class ›face‹ (21.28&#160;%) dominates
                        more semantically charged labels, followed by ›Mary‹ (7.46&#160;%) and ›Baby
                        Jesus‹ (6.04&#160;%). Attributes critical for saint identification&#160;–
                        such as the ›lily‹ (0.92&#160;%)&#160;– are relegated to statistical noise,
                        with over half of the classes represented by fewer than 50 instances. The
                        implications are methodological as much as statistical. Evaluating these
                        corpora has the potential to reward models for recognizing abundant, generic
                        classes while obscuring failures in categories of genuine iconographic
                        expertise. In other words, a model may perform convincingly on aggregate
                        measures while exhibiting no real grasp of the iconographic logics central
                        to art-historical interpretation. However, both IconArt and ArtDL are
                        indispensable precisely because they are the only datasets that even begin
                        to capture iconographic detail. They are best understood, then, not as
                        definitive benchmarks but as starting points&#160;– ›partial ground-truths‹,
                        so to speak, that expose, as much as they enable, the epistemic challenges
                        of employing VLMs in art history.</p>
                    <table xml:id="tab_001">
                        <row role="label">
                            <cell rows="2">Dataset</cell>
                            <cell cols="3">Images</cell>
                            <cell rows="2">Boxes</cell>
                            <cell rows="2">Classes</cell>
                        </row>
                        <row role="label">
                            <cell>Total</cell>
                            <cell>Positive</cell>
                            <cell>Negative</cell>
                        </row>
                        <row>
                            <cell role="label">ArtDL</cell>
                            <cell>4,166</cell>
                            <cell>808</cell>
                            <cell>3,358</cell>
                            <cell>3,793</cell>
                            <cell>59</cell>
                        </row>
                        <row>
                            <cell role="label">IconArt</cell>
                            <cell>1,480</cell>
                            <cell>1037</cell>
                            <cell>443</cell>
                            <cell>4,931</cell>
                            <cell>10</cell>
                        </row>
                        <trailer><ref type="intern" target="#tab1">Table&#160;1</ref>: Statistics of
                            the ArtDL (<ref type="bibliography"
                                target="#milani_fraternali_dataset_2021">Milani&#160;/ Fraternali
                                2021</ref>) and IconArt (<ref type="bibliography"
                                target="#gonthier_et_al_computervision_2018">Gonthier et&#160;al.
                                2018</ref>) test datasets. Positive images contain at least one ground-truth annotation; negative images contain none.</trailer>
                    </table>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_003" url="Medien/vlm_ArtHistory_003.png">
                            <desc>
                                <ref type="intern" target="#abb3">Figure&#160;3</ref>: Selected
                                images are shown from the IconArt (top row; <ref type="bibliography"
                                    target="#gonthier_et_al_computervision_2018">Gonthier
                                    et&#160;al. 2018</ref>) and ArtDL (bottom row; <ref
                                    type="bibliography" target="#milani_fraternali_dataset_2021"
                                    >Milani&#160;/ Fraternali 2021</ref>) test sets. [Visualization:
                                Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                </div>
                <div type="subchapter">
                    <head>3.3 Results</head>
                    <p>
                        <ref type="graphic" target="#tab_002">Table 2</ref> reports the comparative
                        performance of all evaluated methods. Across both the IconArt and ArtDL test
                        sets, CLIP Surgery achieved higher accuracy scores than all other methods,
                        particularly at the more permissive IoU threshold of 0.30. On the ArtDL test
                        set, it obtained a BoxAcc of 52.28&#160;% at an IoU threshold of 0.30&#160;–
                        an absolute improvement of almost 9 points over the second-best method,
                        LeGrad (43.82&#160;%); this is also evident at the stricter IoU threshold of
                        0.50, where CLIP Surgery yielded a BoxAcc of 30.19&#160;%. This advantage
                        extended across object scales:<note type="footnote">Objects were grouped according to the area of their ground-truth bounding box relative to the total image area into small, medium, and large categories.</note> At IoU ≥ 0.30, CLIP Surgery demonstrated the
                        highest accuracy for small (20.69&#160;%), medium (49.46&#160;%), and large
                        objects (74.87&#160;%). We observed some exceptions at the class
                        level&#160;– for instance, ›baby Jesus‹ performed better with LeGrad
                        (76.79&#160;%) than with CLIP Surgery (48.21&#160;%)&#160;– but these are
                        relatively isolated cases. Even when the threshold rose to IoU ≥ 0.50, where
                        precise localization is more challenging, it remained the most accurate in
                        all size categories. In contrast, gradient-based methods&#160;– Grad-CAM,
                        Grad-CAM++, and LayerCAM&#160;– experienced significant performance
                        degradation under both IoU thresholds. On the IconArt test set, CLIP Surgery
                        similarly achieved the highest BoxAcc at IoU ≥ 0.30 (28.76&#160;%); yet
                        LeGrad slightly outperformed it in medium-object accuracy (25.85&#160;%
                        versus 24.47&#160;%). In addition, object-level analysis revealed
                        considerable differences between the two leading methods. For instance, in
                        IconArt, CLIP Surgery recognized the class ›Mary‹ with an accuracy of
                        85.82&#160;% for large objects, compared to 76.62&#160;% with LeGrad. The
                        difference is even greater for ›nudity‹, where CLIP Surgery attained
                        66.16&#160;% versus LeGrad’s 50.25&#160;%. At the stricter IoU ≥ 0.50, CLIP
                        Surgery reclaimed overall superiority (with a BoxAcc of 14.82&#160;%), with
                        LeGrad performing marginally better on small and medium objects.
                        Gradient-based methods performed still more poorly here, underscoring their
                        limited transferability to iconographic material.</p>
                    <table xml:id="tab_002">
                        <row role="label">
                            <cell rows="2">Dataset</cell>
                            <cell rows="2">Method</cell>
                            <cell cols="4">IoU ≥ 0.30</cell>
                            <cell cols="4">IoU ≥ 0.50</cell>
                        </row>
                        <row role="label">
                            <cell>BoxAcc</cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">S</hi></hi></cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">M</hi></hi></cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">L</hi></hi></cell>
                            <cell>BoxAcc</cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">S</hi></hi></cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">M</hi></hi></cell>
                            <cell>BoxAcc<hi rend="sub"><hi rend="italic">L</hi></hi></cell>
                        </row>
                        <row>
                            <cell role="label" rows="7">IconArt</cell>
                            <cell>CLIP Surgery</cell>
                            <cell><hi rend="bold">0.2876</hi></cell>
                            <cell><hi rend="bold">0.0862</hi></cell>
                            <cell>0.2447</cell>
                            <cell><hi rend="bold">0.6623</hi></cell>
                            <cell><hi rend="bold">0.1482</hi></cell>
                            <cell>0.0219</cell>
                            <cell>0.0807</cell>
                            <cell><hi rend="bold">0.4070</hi></cell>
                        </row>
                        <row>
                            <cell>LeGrad</cell>
                            <cell>0.2722</cell>
                            <cell>0.0849</cell>
                            <cell><hi rend="bold">0.2585</hi></cell>
                            <cell>0.6436</cell>
                            <cell>0.1369</cell>
                            <cell><hi rend="bold">0.0232</hi></cell>
                            <cell><hi rend="bold">0.0981</hi></cell>
                            <cell>0.3808</cell>
                        </row>
                        <row>
                            <cell>ScoreCAM</cell>
                            <cell>0.2411</cell>
                            <cell>0.0433</cell>
                            <cell>0.2264</cell>
                            <cell>0.6230</cell>
                            <cell>0.1040</cell>
                            <cell>0.0107</cell>
                            <cell>0.0852</cell>
                            <cell>0.2953</cell>
                        </row>
                        <row>
                            <cell>gScoreCAM</cell>
                            <cell>0.2344</cell>
                            <cell>0.0679</cell>
                            <cell>0.2356</cell>
                            <cell>0.6024</cell>
                            <cell>0.1121</cell>
                            <cell>0.0174</cell>
                            <cell>0.0843</cell>
                            <cell>0.3208</cell>
                        </row>
                        <row>
                            <cell>GradCAM</cell>
                            <cell>0.1391</cell>
                            <cell>0.0666</cell>
                            <cell>0.2145</cell>
                            <cell>0.2509</cell>
                            <cell>0.0355</cell>
                            <cell>0.0156</cell>
                            <cell>0.0660</cell>
                            <cell>0.0624</cell>
                        </row>
                        <row>
                            <cell>GradCAM++</cell>
                            <cell>0.1584</cell>
                            <cell>0.0277</cell>
                            <cell>0.0880</cell>
                            <cell>0.4432</cell>
                            <cell>0.0584</cell>
                            <cell>0.0067</cell>
                            <cell>0.0192</cell>
                            <cell>0.1779</cell>
                        </row>
                        <row>
                            <cell>LayerCAM</cell>
                            <cell>0.1783</cell>
                            <cell>0.0420</cell>
                            <cell>0.1769</cell>
                            <cell>0.4694</cell>
                            <cell>0.0627</cell>
                            <cell>0.0094</cell>
                            <cell>0.0577</cell>
                            <cell>0.1823</cell>
                        </row>
                        <row>
                            <cell role="label" rows="7">ArtDL</cell>
                            <cell>CLIP Surgery</cell>
                            <cell><hi rend="bold">0.5228</hi></cell>
                            <cell><hi rend="bold">0.2069</hi></cell>
                            <cell><hi rend="bold">0.4946</hi></cell>
                            <cell><hi rend="bold">0.7487</hi></cell>
                            <cell><hi rend="bold">0.3019</hi></cell>
                            <cell><hi rend="bold">0.0722</hi></cell>
                            <cell><hi rend="bold">0.2205</hi></cell>
                            <cell><hi rend="bold">0.5297</hi></cell>
                        </row>
                        <row>
                            <cell>LeGrad</cell>
                            <cell>0.4382</cell>
                            <cell>0.1788</cell>
                            <cell>0.4280</cell>
                            <cell>0.7035</cell>
                            <cell>0.2552</cell>
                            <cell>0.0458</cell>
                            <cell>0.1868</cell>
                            <cell>0.4680</cell>
                        </row>
                        <row>
                            <cell>ScoreCAM</cell>
                            <cell>0.3557</cell>
                            <cell>0.0667</cell>
                            <cell>0.3055</cell>
                            <cell>0.6457</cell>
                            <cell>0.1672</cell>
                            <cell>0.0125</cell>
                            <cell>0.1087</cell>
                            <cell>0.3441</cell>
                        </row>
                        <row>
                            <cell>gScoreCAM</cell>
                            <cell>0.3815</cell>
                            <cell>0.0889</cell>
                            <cell>0.3675</cell>
                            <cell>0.6627</cell>
                            <cell>0.1727</cell>
                            <cell>0.0208</cell>
                            <cell>0.1378</cell>
                            <cell>0.3497</cell>
                        </row>
                        <row>
                            <cell>GradCAM</cell>
                            <cell>0.2684</cell>
                            <cell>0.1056</cell>
                            <cell>0.3851</cell>
                            <cell>0.2790</cell>
                            <cell>0.0701</cell>
                            <cell>0.0278</cell>
                            <cell>0.0988</cell>
                            <cell>0.0787</cell>
                        </row>
                        <row>
                            <cell>GradCAM++</cell>
                            <cell>0.2418</cell>
                            <cell>0.0403</cell>
                            <cell>0.1493</cell>
                            <cell>0.4890</cell>
                            <cell>0.1168</cell>
                            <cell>0.0069</cell>
                            <cell>0.0237</cell>
                            <cell>0.2484</cell>
                        </row>
                        <row>
                            <cell>LayerCAM</cell>
                            <cell>0.2476</cell>
                            <cell>0.0583</cell>
                            <cell>0.2052</cell>
                            <cell>0.4578</cell>
                            <cell>0.0873</cell>
                            <cell>0.0097</cell>
                            <cell>0.0582</cell>
                            <cell>0.1590</cell>
                        </row>
                        <trailer><ref type="intern" target="#tab2">Table&#160;2</ref>: Bounding box
                            detection results are reported for the IconArt (<ref type="bibliography"
                                target="#gonthier_et_al_computervision_2018">Gonthier et al.
                                2018</ref>) and ArtDL (<ref type="bibliography"
                                target="#milani_fraternali_dataset_2021">Milani&#160;/ Fraternali
                                2021</ref>) test sets, using the highest BoxAcc obtained across all
                            binarization thresholds τ. The best performing approach per test set is
                            indicated in bold. The subscripts S, M, and L denote ›small‹, ›medium‹, and ›large‹ objects, respectively.</trailer>
                    </table>
                    <p>The lower overall accuracy observed on IconArt relative to ArtDL arises from
                        both structural and semantic differences between the two corpora. First,
                        IconArt contains a higher proportion of small objects&#160;– those occupying
                        ≤ 1&#160;% of the image area&#160;– with 46.30&#160;% of all instances,
                        versus 18.98&#160;% in ArtDL. Small objects are inherently more challenging
                        to detect and classify, because they provide fewer pixels for feature
                        extraction and are more susceptible to background clutter; the
                        overrepresentation of small objects consequently reduces the aggregate
                        performance on IconArt. Second, the datasets have different epistemic
                        scopes. ArtDL is broader and more generic, including categories that can be
                        recognized without specialized iconographic knowledge, e.g., ›beard‹.
                        IconArt, by contrast, focuses on a small number of historically charged
                        motifs&#160;– such as the ›Crucifixion of Jesus‹&#160;– whose correct
                        identification depends on contextual and narrative cues rather than isolated
                        attributes. These scenes are formally and semantically dense: Their
                        complexity and entanglement with related subthemes (e.g., episodes from
                        Christ’s Passion) expose the limitations of models optimized for visual
                        generality.</p>
                </div>
            </div>
            <div type="chapter">
                <head>4. Case Study 2</head>
                <p>So, what can we establish at this point? Across two large-scale datasets, our
                    evaluation shows that CLIP Surgery consistently outperformed all other methods,
                    with LeGrad emerging&#160;– almost unequivocally&#160;– as the second-best
                    approach. Yet the discrepancies between the two datasets, IconArt and ArtDL, are
                    instructive. While both annotate iconographic figures and attributes, they,
                    inevitably, only partially reflect the broader iconographic heterogeneity that
                    defines art-historical material. Our large-scale approach therefore provides
                    valuable, albeit ultimately limited, perspectives as the selected classes
                    naturally constrain the horizon of possible inquiry: We can assess performance
                    within these constraints, but we cannot assume that the epistemic space of art
                    history is adequately represented.<note type="footnote">This limitation, of
                        course, applies to all machine-vision research, since available
                        datasets&#160;– shaped by resource constraints&#160;– necessarily offer a
                        highly selective view of historical material tailored to specific research
                        questions.</note> The second case study thus deliberately shifted focus. It
                    addressed the gap already identified by Miller, namely the absence of
                    human-centered evaluations of XAI methods.<note type="footnote">Cf. <ref
                            type="bibliography" target="#miller_explanation_2019">Miller
                        2019</ref>.</note> While the first case study focused on measuring the
                    accuracy of these methods and assigning numerical performance scores, this
                    second case study explored <hi rend="italic">interpretability</hi>&#160;– asking
                    not only <hi rend="italic">what</hi> these models predict, but <hi rend="italic"
                        >how</hi> their explanations were understood by human users. To approximate
                    the practical variability of art-historical research&#160;– resistant as it is
                    to fixed classification&#160;– we employed a broader and more diverse selection
                    of images and categories. Our methodological scope remained consistent with the
                    first study, encompassing the same explanation methods&#160;– Grad-CAM,
                    Grad-CAM++, LayerCAM, LeGrad, ScoreCAM, gScoreCAM, and CLIP Surgery. But the
                    focus was no longer on algorithmic performance. Rather, it tested whether these
                    methods could make visible the complex visual logics that art history engages
                    with, and whether they succeed, or fail, in approximating the visual concepts
                    that constitute the field.</p>
                <div type="subchapter">
                    <head>4.1 Experimental Design</head>
                    <p>To assess the degree to which algorithm-generated saliency maps align with
                        human perceptions of visual importance, we conducted a within-subjects
                        online study using <ref target="https://www.soscisurvey.de/">SoSci
                            Survey</ref> between June and July 2025.<note type="footnote">The survey
                            was last accessed on October 10, 2025.</note> After reviewing and
                        accepting an informed consent form, participants provided socio-demographic
                        information, including age, gender identity, education level, and
                        professional status. On each subsequent page, they were shown one of seven
                        artworks and asked to use their mouse to annotate regions they deemed
                        relevant to a specified class (<ref type="graphic"
                            target="#vlm_ArtHistory_004">Fig.&#160;4</ref>). These annotations served as the human reference (›ground-truth‹) for the subsequent evaluation. Each artwork was paired
                        with two target classes: Petrus Christus’s <title>A Goldsmith in his
                            Shop</title> (1449) with ›convex mirror‹ and ›girdle‹; Franz von Stuck’s
                            <title>Adam and Eve</title> (c. 1920) with ›arm outstretched‹ and
                        ›snake‹; Antonello da Messina’s <title>Calvary</title> (1475) with ›John‹
                        and ›thief‹; Claude Monet’s <title>Japanese Footbridge</title> (1899) with
                        ›bridge‹ and ›flower‹; Jean-Auguste-Dominique Ingres’s <title>Oedipus and
                            the Sphinx</title> (1808) with ›left foot‹ and ›Sphinx‹; Sandro
                        Botticelli’s <title>The Lamentation</title> (c. 1490) with ›sword‹ and
                        ›Virgin Mary‹; Bartholomeus van der Helst’s <title>The Musician</title>
                        (1662) with ›lustful‹ and ›sheet music‹. The selection of artworks spanned a
                        broad range of periods and styles, from Renaissance devotional works to
                        early twentieth-century Symbolism. This diversity was intentional: It
                        ensured that annotators were exposed to heterogeneous visual traditions and
                        compositional strategies. Some target classes referred to discrete, visually
                        localized elements (e.g., ›bridge‹), while others invoked symbolic or
                        abstract categories (e.g., ›lustful‹). The participants then viewed seven
                        saliency maps for each image–class pair, again generated by the following
                        explanation methods: Grad-CAM, Grad-CAM++, LayerCAM, LeGrad, ScoreCAM,
                        gScoreCAM, and CLIP Surgery. They were asked to order these maps according
                        to how well they reflected the regions they had previously identified as
                        important. This ranking served as a subjective measure of the alignment
                        between human visual attention and the algorithmic saliency outputs. To
                        minimize order effects, both the sequence of image-class pairs and the
                        presentation order of saliency maps were independently randomized for each
                        participant. Only those participants who completed at least 4 of the 14
                        ranking tasks were included in the final analysis.</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_004" url="Medien/vlm_ArtHistory_004.png">
                            <desc>
                                <ref type="intern" target="#abb4">Figure&#160;4</ref>: Artworks
                                selected for the online study: Petrus Christus, <title>A Goldsmith
                                    in his Shop</title> (1449; a); Franz von Stuck, <title>Adam and
                                    Eve</title> (c. 1920; b); Antonello da Messina,
                                    <title>Calvary</title> (1475; c); Claude Monet, <title>Japanese
                                    Footbridge</title> (1899; d); Jean-Auguste-Dominique Ingres,
                                    <title>Oedipus and the Sphinx</title> (1808; e); Sandro
                                Botticelli, <title>The Lamentation</title> (c. 1490; f);
                                Bartholomeus van der Helst, <title>The Musician</title> (1662; g).
                                [Visualization: Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                </div>
                <div type="subchapter">
                    <head>4.2 Participants</head>
                    <p>Participants were recruited through university-based channels at the University of
                        Munich and the University of Göttingen; participation was voluntary and not
                        incentivized. The study involved 33 participants, of whom 21.21&#160;%
                        identified as male, 75.76&#160;% as female, and 3.03&#160;% as diverse.
                        Participant ages ranged from 18 to over 65 years (mean age = 42.21 years; standard deviation = 19.08 years).
                        Regarding educational background, 39.39&#160;% held qualifications for
                        university entry, while 45.46&#160;% had completed a degree at either a
                        university or a university of applied sciences. Students made up the largest
                        subgroup, accounting for 54.54&#160;% of all respondents. In terms of
                        art-historical expertise, 62.50&#160;% reported basic knowledge (e.g.,
                        gained through introductory coursework), 21.88&#160;% reported intermediate
                        proficiency (gained through a bachelor’s degree or initial professional
                        experience), and the remainder described themselves as advanced or expert.
                        The demographic profile thus reflects a fairly diverse sample in terms of
                        both age and educational attainment. At the participant level, a
                        complete-case analysis would have excluded 37.93&#160;% of participants due
                        to incomplete rankings for one or more image-class pairs; at the level of
                        individual ranking tasks, 8.86&#160;% were incomplete because not all seven
                        methods had been ranked. We therefore imputed missing rank positions using
                        the <term type="dh">Multivariate Imputation by Chained Equations
                            (MICE)</term> algorithm over 20 iterations.<note type="footnote">Cf.
                                <ref type="bibliography"
                                target="#buuren_groothuis-oudshoorn_mice_2011">Buuren&#160;/
                                Groothuis-Oudshoorn 2011</ref>.</note> Although the sample size of n
                        = 33 is at the lower end of recommended thresholds for within-subjects
                        tests, power analysis indicates sufficient sensitivity to detect medium
                        levels of inter-rater agreement (Kendall’s coefficient of concordance W ≥
                        0.30) with an estimated probability of ≈ 80–85&#160;%. For the purposes of
                        this investigation, such values are deemed adequate: The study was not
                        intended to establish statistically significant differences between these
                        specific artworks, but rather to assess&#160;– at least to some
                        extent&#160;– the general ranking of saliency maps in terms of their
                        alignment with human annotations. The central objective was to determine
                        whether a modest, curated sample obtained through a user study could already
                        reveal broader patterns that are informative for evaluating the
                        applicability of XAI methods in art-historical contexts.</p>
                </div>
                <div type="subchapter">
                    <head>4.3 Results</head>
                    <p>
                        <ref type="graphic" target="#vlm_ArtHistory_005">Fig.&#160;5</ref> shows
                        divergent stacked bar charts of ranking distributions for each image–class
                        pair. The charts illustrate the extent to which participants agreed or
                        disagreed in their assessments, making it easier to identify patterns of
                        consensus or divergence across images. To formally quantify inter-rater
                        agreement, we report <term type="dh">Kendall’s coefficient of concordance (W)</term> for each image–class pair: It ranges from 0 to 1, with higher values
                        indicating stronger reliability and lower values greater variability in judgments. Across the images, three techniques&#160;– CLIP
                        Surgery, LeGrad, and ScoreCAM&#160;– achieve the highest mean rank
                        positions, indicating that participants perceived these methods as most
                        faithfully highlighting their annotated regions; gScoreCAM also performs
                        strongly, but shows slightly greater variability. In contrast, Grad-CAM,
                        Grad-CAM++, and LayerCAM are consistently ranked towards the bottom,
                        suggesting that their gradient-based heatmaps align less closely with the
                        participants’ own saliency judgments.</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_005" url="Medien/vlm_ArtHistory_005.png">
                            <desc>
                                <ref type="intern" target="#abb5">Figure&#160;5</ref>: Evaluation
                                results are shown as divergent stacked bar charts, comparing seven
                                visual explainability methods for each image and class. Colors range
                                from red (›least accurate‹) to blue (›most accurate‹). Kendall’s W
                                is reported to assess inter-rater reliability. [Visualization:
                                Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_006" url="Medien/vlm_ArtHistory_006.png">
                            <desc>
                                <ref type="intern" target="#abb6">Figure&#160;6</ref>: Evaluation
                                results are shown as divergent stacked bar charts, comparing seven
                                XAI methods across different levels of art-historical expertise. The
                                results are aggregated over all images and classes. Colors range
                                from red (›least accurate‹) to blue (›most accurate‹).
                                [Visualization: Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <p>When aggregated across all images and classes, these patterns were largely
                        consistent across levels of art-historical expertise, as shown in <ref
                            type="graphic" target="#vlm_ArtHistory_006">Fig.&#160;6</ref>. While
                        CLIP Surgery was favored by participants with basic knowledge, those with at
                        least intermediate proficiency had a slight preference for LeGrad; at this
                        level, gScoreCAM also approached the top-performing group. Beyond these
                        minor differences, no further&#160;– and in particular, no statistically
                        significant&#160;– effects are observed. For visually well-defined or
                        spatially localized targets&#160;– e.g., the ›snake‹ in Franz von Stuck’s
                            <title>Adam and Eve</title> (<ref type="graphic"
                            target="#vlm_ArtHistory_007">Fig.&#160;7</ref>) or the ›left foot‹ in
                        Jean-Auguste-Dominique Ingres’s <title>Oedipus and the Sphinx</title> (<ref
                            type="graphic" target="#vlm_ArtHistory_008">Fig.&#160;8</ref>)&#160;–
                        participants’ rankings converge tightly, with Kendall’s W indicating strong
                        inter-rater reliability (W = 0.71 and W = 0.62, respectively).</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_007" url="Medien/vlm_ArtHistory_007.png">
                            <desc>
                                <ref type="intern" target="#abb7">Figure&#160;7</ref>: Ground-truth
                                annotations and saliency maps for Franz von Stuck’s <title>Adam and
                                    Eve</title> (c. 1920), shown for the classes ›arm outstretched‹
                                (top row) and ›snake‹ (bottom row). [Visualization: Stefanie
                                Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_008" url="Medien/vlm_ArtHistory_008.png">
                            <desc>
                                <ref type="intern" target="#abb8">Figure&#160;8</ref>: Ground-truth
                                annotations and saliency maps for Jean-Auguste-Dominique Ingres’s
                                    <title>Oedipus and the Sphinx</title> (1808), shown for the
                                classes ›left foot‹ (top row) and ›Sphinx‹ (bottom row).
                                [Visualization: Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <p>A similar pattern emerges for the flowers beneath the bridge in Claude
                        Monet’s painting (W = 0.83), pointing to a shared perception of salient
                        features (<ref type="graphic" target="#vlm_ArtHistory_009"
                        >Fig.&#160;9</ref>). By contrast, more diffuse or interpretative categories,
                        such as ›lustful‹ (<ref type="graphic" target="#vlm_ArtHistory_010"
                            >Fig.&#160;10</ref>) or the ›Sphinx‹ (<ref type="graphic"
                            target="#vlm_ArtHistory_008">Fig.&#160;8</ref>), yield widely dispersed
                        rankings and low values Kendall’s W, with no single method emerging as
                        dominant; this reflects the intrinsic ambiguity of these higher-order visual
                        concepts. However, the difficulty of some classes also becomes evident in
                        cases where even human annotators struggle to establish consistent
                        associations. Take, for example, Ingres’s <title>Oedipus and the
                            Sphinx</title>, in which three distinct ›left feet‹ would, in principle,
                        need to be annotated: (1) The left foot of Oedipus, firmly bracing his
                        weight; (2) the left foot of the Sphinx, obscured in shadow and therefore
                        difficult to recognize&#160;– mirrored by the sparse human annotations in
                            <ref type="graphic" target="#vlm_ArtHistory_008">Fig.&#160;8</ref>, and
                        by the failure of saliency maps to capture it; and (3) a severed left foot
                        at the lower edge of the canvas, possibly belonging to one of the Sphinx’s
                        earlier victims. In this case, annotators likely registered only the most
                        prominent instance, suggesting insufficient time for close observation.</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_009" url="Medien/vlm_ArtHistory_009.png">
                            <desc>
                                <ref type="intern" target="#abb9">Figure&#160;9</ref>: Ground-truth
                                annotations and saliency maps for Claude Monet’s <title>Japanese
                                    Footbridge</title> (1899), shown for the classes bridge‹ (top
                                row) and ›flower‹ (bottom row). [Visualization: Stefanie Schneider
                                2026]</desc>
                        </graphic>
                    </figure>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_010" url="Medien/vlm_ArtHistory_010.png">
                            <desc>
                                <ref type="intern" target="#abb10">Figure&#160;10</ref>:
                                Ground-truth annotations and saliency maps for Bartholomeus van der
                                Helst’s <title>The Musician</title> (1662), shown for the classes
                                ›lustful‹ (top row) and ›sheet music‹ (bottom row). [Visualization:
                                Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <p>Yet in other examples, the challenge arose less from attention than from
                        limited knowledge of the relevant visual structures associated with a given
                        class. An interesting case is the red wedding girdle in Petrus Christus’s
                            <title>Goldsmith</title>, which projects into the viewer’s space over
                        the shop’s ledge, yet is far less frequently annotated than the woman’s hip
                        belt (<ref type="graphic" target="#vlm_ArtHistory_011">Fig.&#160;11</ref>).
                        The issue is even more evident in Sandro Botticelli’s
                            <title>Lamentation</title>. While the dead Christ rests in the Virgin
                        Mary’s lap, two additional figures&#160;– Mary Magdalene and, very likely,
                        Mary of Clopas&#160;– gently support his head and feet. Yet participants
                        frequently mislabeled these figures as the Virgin Mary as well (<ref
                            type="graphic" target="#vlm_ArtHistory_012">Fig.&#160;12</ref>).</p>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_011" url="Medien/vlm_ArtHistory_011.png">
                            <desc>
                                <ref type="intern" target="#abb11">Figure&#160;11</ref>:
                                Ground-truth annotations and saliency maps for Petrus Christus’s
                                    <title>A Goldsmith in his Shop</title> (1449), shown for the
                                classes ›convex mirror‹ (top row) and ›girdle‹ (bottom row).
                                [Visualization: Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                    <figure>
                        <graphic xml:id="vlm_ArtHistory_012" url="Medien/vlm_ArtHistory_012.png">
                            <desc>
                                <ref type="intern" target="#abb12">Figure&#160;12</ref>:
                                Ground-truth annotations and saliency maps for Sandro Botticelli’s
                                    <title>The Lamentation</title> (c. 1490), shown for the classes
                                ›sword‹ (top row) and ›Virgin Mary‹ (bottom row). [Visualization:
                                Stefanie Schneider 2026]</desc>
                        </graphic>
                    </figure>
                </div>
            </div>
            <div type="chapter">
                <head>5. Discussion</head>
                <p>The case studies reveal two interrelated factors. (1) <hi rend="italic">Ambiguity
                        of concepts</hi>: In art-historical imagery, target classes often are
                    interpretive rather than fixed, and thus resist stable localization. In
                    Botticelli’s <title>Lamentation</title>, the three Marys mourning over Christ
                    are visually similar, so that non-specialist annotators might confuse them (<ref
                        type="graphic" target="#vlm_ArtHistory_012">Fig.&#160;12</ref>). Here, <hi
                        rend="italic">discernibility</hi> emerges as a criterion: Where classes are
                    both concrete and spatially bounded, annotators converge; where they are
                    symbolic, context-dependent, or reliant on art-historical expertise, judgments
                    diverge and inter-rater reliability declines. Ground-truth annotations, in this
                    sense, are never exhaustive. Methods that highlight only the most visually
                    dominant instance of a class therefore cannot easily be dismissed as inadequate,
                    since they nonetheless establish a valid link between text and image. Yet
                    ambiguity does not arise only from annotators: It is also encoded in the model
                    itself.</p>
                <p>This leads to: (2) <hi rend="italic">Limits of representation</hi>: Saliency
                    methods cannot recover what the model itself fails to encode. If a concept does
                    not appear as a localized hotspot in CLIP’s latent space, attribution will
                    necessarily remain vague, regardless of the post-hoc technique employed. This is
                    evident in Antonello da Messina’s <title>Calvary</title>: The thieves flanking
                    Christ are inconsistently mapped&#160;– ScoreCAM and gScoreCAM weakly associate
                    both figures with the prompt ›thief‹, while LeGrad isolates only the feet of the
                    right-hand figure (<ref type="graphic" target="#vlm_ArtHistory_013"
                        >Fig.&#160;13</ref>). Such incoherence suggests that CLIP does not encode
                    ›thief‹ as a transferable visual concept. The difficulty may partly stem from
                    the unnatural posture of the figures, but&#160;– more fundamentally&#160;– it
                    reflects the fragmentary nature of the training corpus. In contrast to the
                    photographic imagery that dominates CLIP’s dataset, crucifixion scenes are
                    relatively uncommon, with peripheral figures such as the thieves being
                    especially under-specified. The challenge is further complicated by the semantic
                    breadth of the term ›thief‹. Unlike ›Christ‹ or ›cross‹, which correspond to
                    highly codified and visually stable iconographic forms, ›thief‹ has no fixed
                    template: It may denote an anonymous criminal, a masked burglar, or&#160;– in
                    the Passion narrative&#160;– the unnamed ›good‹ and ›bad‹ thieves, who are
                    distinguished only by their position relative to Christ and, in some traditions,
                    by subtle gestures or expressions. Ambiguity, in this case, is not simply a
                    problem of human perception, but a structural property of the machine’s
                    representational logic. Yet across both cases, relative performance of
                    gradient-based, score-based, and hybrid methods remained largely stable.</p>
                <p>This indicates that small-scale, human-centered studies can support robust
                    comparative evaluation, while large-scale evaluation on pre-existing datasets
                    can extend such results&#160;– though only for a narrow subset of
                    art-historically relevant categories. More importantly, these studies also
                    function diagnostically: They reveal not only how well attribution methods
                    capture model activations, but also how models themselves mediate art-historical
                    concepts. Human studies, carefully designed, thus can produce findings that
                    generalize to broader validations while simultaneously exposing the cultural and
                    epistemic imaginaries embedded in machine vision.</p>
                <figure>
                    <graphic xml:id="vlm_ArtHistory_013" url="Medien/vlm_ArtHistory_013.png">
                        <desc>
                            <ref type="intern" target="#abb13">Figure&#160;13</ref>: Ground-truth
                            annotations and saliency maps for Antonello da Messina’s
                                <title>Calvary</title> (1475), shown for the classes ›John‹ (top
                            row) and ›thief‹ (bottom row). [Visualization: Stefanie Schneider
                            2026]</desc>
                    </graphic>
                </figure>
                <p>For real-time or ad hoc explanation needs, additional considerations are
                    required. ScoreCAM obtains channel weights by computing the forward-pass score
                    for each activation map&#160;– i.e., it performs one forward pass per channel
                    (i.e., C forward passes per image)&#160;– to generate a single heatmap (e.g.,
                    3,072 forward passes for RN50×16), which makes it slow for on-the-fly
                        explanations.<note type="footnote">Cf. <ref type="bibliography"
                            target="#wang_et_al_score-CAM_2020">Wang et&#160;al. 2020</ref>.</note>
                    gScoreCAM alleviates this bottleneck by selecting only the top-k channels
                    (typically k = 300), reducing the number of forward passes to about 0.1C ≈ 307
                    for C = 3,072.<note type="footnote">Cf. <ref type="bibliography"
                            target="#chen_et_al_gScoreCAM_2022">Chen et&#160;al. 2022</ref>.</note>
                    By contrast, gradient-based methods (Grad-CAM, Grad-CAM++, LayerCAM, and
                    variants such as LeGrad) require one forward pass and one backward pass to
                    compute CAMs. However, for real-time or ad hoc explanations, CLIP Surgery is
                    particularly advantageous, as it requires only a single modified forward pass
                    without any gradient computations or backpropagation.<note type="footnote">Cf.
                            <ref type="bibliography" target="#li_et_al_explainability_2025">Li
                            et&#160;al. 2025</ref>. Actual runtime performance depends on hardware
                        and implementation details. We recommend empirical benchmarking of all
                        methods on the target system to obtain accurate latency estimates before
                        deployment.</note> Beyond computational efficiency, these methodological
                    differences also influence the epistemic character of the resulting
                    explanations. Multi-pass methods like ScoreCAM and gScoreCAM often produce
                    smoother, less noisy saliency maps, but their cost makes them impractical for
                    interactive settings. Gradient-based methods, while faster, can suffer from
                    gradient saturation or overemphasis on high-level features, resulting in
                    instability or sensitivity across inputs. CLIP Surgery, though extremely
                    efficient, forgoes gradient information entirely; its outputs should thus be
                    interpreted as approximations of model ›attention‹ rather than as exhaustive
                    mappings of activation contributions. In art-historical contexts, this
                    distinction is nontrivial: When interpreting ambiguous or symbolical imagery,
                    the choice of method not only constrains computational performance but also
                    shapes the interpretive claims about what the model <hi rend="italic"
                        >perceives</hi>. Real-time explanation pipelines, then, must balance latency
                    constraints with the need for stable, interpretable visualizations&#160;–
                    accepting that faster methods may favor responsiveness and accessibility, while
                    slower methods may deliver higher resolution or fidelity but at the cost of
                    usability in exploratory settings.</p>
            </div>
            <div type="chapter">
                <head>6. Conclusion</head>
                <p>Our case studies&#160;– addressing the questions raised in <ref type="intern"
                        target="#hd1">Section 1</ref>&#160;– demonstrated that the epistemic promise
                    of XAI in digital art history is both methodological and hermeneutic. With
                    regard to the first question (<hi rend="italic">How effectively do XAI methods
                        localize iconographic objects in artworks under zero-shot conditions without
                        fine-tuning?</hi>), our quantitative evaluation has shown that CLIP Surgery,
                    which is specifically adapted to CLIP’s dual-encoder architecture, consistently
                    outperformed general-purpose gradient- or score-based methods. Even when trained
                    primarily on non-art-historical imagery, CLIP Surgery could delineate a broad
                    range of iconographic concepts, particularly when the target classes are
                    visually distinct or spatially constrained; its accuracy, however, declined for
                    semantically complex motifs. The limitation here lies not in the XAI method
                    itself but in the representational granularity of the embedding space: CLIP does
                    not, in fact, <hi rend="italic">see</hi> an object in its historicity, but only
                    the statistical residue of an already mediated image-world. Turning to the
                    second question (<hi rend="italic">Does the visual relevance of these maps
                        correspond to human judgments?</hi>), our user study partially confirmed
                    this observation. Participants generally preferred CLIP Surgery, LeGrad, and
                    ScoreCAM, whose saliency maps closely approximated human annotations. Yet for
                    abstract or context-dependent classes, agreement and, thus, inter-rater
                    reliability decreased. This divergence underscores a central tension: While
                    saliency maps can reproduce certain aspects of perceptual attention, they cannot
                    replicate the interpretive depth of the art-historical gaze. The third question
                        (<hi rend="italic">Which factors, such as object size and concept
                        abstraction, drive performance differences?</hi>) reveals both structural
                    and semantic determinants. Localization accuracy correlates strongly with object
                    size&#160;– and, thus, visual prominence&#160;– but it is equally influenced by
                    conceptual stability. Classes with more consistent visual referents (e.g.,
                    ›bridge‹ or ›snake‹) yielded higher accuracy scores than those with vague,
                    symbolic, or art-historically specific meanings (e.g., ›lustful‹ or ›Virgin
                    Mary‹).</p>
                <p>These findings suggest that model performance reflects how, and to what extent, a
                    concept exists within the model’s learned visual ontology. Returning to a
                    broader question raised in <ref type="intern" target="#hd1">Section
                    1</ref>&#160;– whether XAI discloses a model’s internal conceptual structure or
                    merely aestheticizes its opacity&#160;– our answer must be: It depends. The
                    visual legibility of a saliency map can be deceptive as it does not imply
                    epistemic transparency. Such maps expose the internal dynamics of how certain
                    features or tokens activate within an embedding space, while concealing the
                    historical, cultural, and linguistic priors that render those activations
                    meaningful. What is visualized, therefore, is not the model’s <hi rend="italic"
                        >understanding</hi> of an artwork but the projection of human interpretive
                    desire onto computational artifacts that can only approximate meaning.
                    Explainability in digital art history must, in this context, be conceived as a
                    dialogical process between human and machine vision&#160;– as elaborated by
                        Miller<note type="footnote">Cf. <ref type="bibliography"
                            target="#miller_explanation_2019">Miller 2019</ref>.</note>&#160;–
                    demanding critical awareness of the epistemic imaginaries through which models
                        <hi rend="italic">see</hi>. XAI outputs should therefore be read not as
                    self-sufficient explanations but as prompts for further hermeneutic inquiry.</p>
            </div>
            <div type="chapter">
                <head>Acknowledgements</head>
                <p>This work was funded in part by the German Research Foundation (Deutsche
                    Forschungsgemeinschaft; DFG) under project number <ref target="https://gepris.dfg.de/project/510048106">510048106</ref>. I thank Hubertus
                    Kohle, Ralph Ewerth, Eric Müller-Budack, Matthias Springstein, and Julian
                    Stalter for their insightful discussions and helpful comments on the subject
                    matter.</p>
            </div>
        </body>
        <back>
            <div type="bibliography">
                <head>Bibliography</head>
                <listBibl>
                    <bibl xml:id="abnar_zidema_attention_2020">Samira Abnar&#160;/ Willem H.
                        Zuidema: Quantifying Attention Flow in Transformers. In: Dan Jurafsky&#160;/
                        Joyce Chai&#160;/ Natalie Schluter&#160;/ Joel R. Tetreault (eds.):
                        Proceedings of the 58<hi rend="super">th</hi> Annual Meeting of the
                        Association for Computational Linguistics (Online, 05.–10.07.2020). Stroudsburg, US-PA 2020, pp. 4190–4197. DOI: <ref target="https://doi.org/10.18653/v1/2020.acl-main.385">10.18653/v1/2020.acl-main.385</ref>
                    </bibl>
                    <bibl xml:id="bender_et_al_language_2021">Emily M. Bender&#160;/ Timnit
                        Gebru&#160;/ Angelina McMillan-Major&#160;/ Shmargaret Shmitchell: On the
                        Dangers of Stochastic Parrots: Can Language Models Be Too Big? In: Madeleine
                        Clare Elish&#160;/ William Isaac&#160;/ Richard S. Zemel (eds.): FAccT ’21:
                        Proceedings of the 2021 ACM Conference on Fairness, Accountability, and
                        Transparency (Canada, 03.03.–10.03.2021). New York 2021, pp.&#160;610–623.
                        DOI: <ref target="https://doi.org/10.1145/3442188.3445922">10.1145/3442188.3445922</ref>
                    </bibl>
                    <bibl xml:id="bhalla_et_al_clip_2024">Usha Bhalla&#160;/ Alex Oesterling&#160;/
                        Suraj Srinivas&#160;/ Flávio P. Calmon&#160;/ Himabindu Lakkaraju:
                        Interpreting CLIP with Sparse Linear Concept Embeddings (SpLiCE). In: Amir
                        Globerson&#160;/ Lester Mackey&#160;/ Danielle Belgrave &#160;/ Angela
                        Fan&#160;/ Ulrich Paquet&#160;/ Jakub M. Tomczak&#160;/ Cheng Zhang (eds.):
                        Advances in Neural Information Processing Systems 37. NeurIPS 2024. 38<hi
                            rend="super">th</hi> Annual Conference on Neural Information Processing
                        Systems (Vancouver, 10.12.–15.12.2024). 2024. [<ref target="https://proceedings.neurips.cc/paper_files/paper/2024/hash/996bef37d8a638f37bdfcac2789e835d-Abstract-Conference.html">online</ref>]</bibl>
                    <bibl xml:id="birhane_et_al_datasets_2021">Abeba Birhane&#160;/ Vinay Uday
                        Prabhu&#160;/ Emmanuel Kahembwe: Multimodal Datasets: Misogyny, Pornography,
                        and Malignant Stereotypes. arXiv. 05.10.2021. DOI: <ref target="https://doi.org/10.48550/arXiv.2110.01963">10.48550/arXiv.2110.01963</ref>
                    </bibl>
                    <bibl xml:id="bousselham_et_al_explainability_2024">Walid Bousselham&#160;/
                        Angie W. Boggust&#160;/ Sofian Chaybouti&#160;/ Hendrik Strobelt&#160;/
                        Hilde Kuehne: LeGrad: An Explainability Method for Vision Transformers via
                        Feature Formation Sensitivity. arXiv. 04.04.2024. Version&#160;2:
                        08.01.2025. DOI: <ref target="https://arxiv.org/abs/2404.03214v2">10.48550/arXiv.2404.03214v2</ref>
                    </bibl>
                    <bibl xml:id="buchanan_shortliffe_eds_expertsystems_1985">Bruce G.
                        Buchanan&#160;/ Edward H. Shortliffe (eds.): Rule-Based Expert Systems: The
                        MYCIN Experiments of the Stanford Heuristic Programming Project. Reading,
                        US-MA etc. 1985. <ptr type="gbv" cRef="221027912"/>
                    </bibl>
                    <bibl xml:id="buuren_groothuis-oudshoorn_mice_2011">Stef van Buuren&#160;/ Karin
                        Groothuis-Oudshoorn: mice: Multivariate Imputation by Chained Equations in
                        R. In: Journal of Statistical Software 45 (2011), no.&#160;3, pp.&#160;1–67.
                        DOI: <ref target="https://doi.org/10.18637/jss.v045.i03">10.18637/jss.v045.i03</ref>
                    </bibl>
                    <bibl xml:id="chattopadhyay_et_al_gradCAM_2018">Aditya Chattopadhyay&#160;/
                        Anirban Sarkar&#160;/ Prantik Howlader&#160;/ Vineeth N. Balasubramanian:
                        Grad-CAM++: Generalized Gradient-Based Visual Explanations for Deep
                        Convolutional Networks. In: 2018 IEEE Winter Conference on Applications of
                        Computer Vision (WACV; Lake Tahoe, US-NV, 12.03.–15.03.2018). 2018,
                        pp.&#160;839–847. DOI: 10.1109/WACV.2018.00097</bibl>
                    <bibl xml:id="chen_et_al_gScoreCAM_2022">Peijie Chen&#160;/ Qi Li&#160;/ Saad
                        Biaz&#160;/ Trung Bui&#160;/ Anh Nguyen: gScoreCAM: What Objects Is CLIP
                        Looking At? In: Lei Wang&#160;/ Juergen Gall&#160;/ Tat-Jun Chin&#160;/
                        Imari Sato&#160;/ Rama Chellappa (eds.): Computer Vision&#160;– ACCV 2022:
                            16<hi rend="super">th</hi> Asian Conference on Computer Vision
                        (=&#160;Lecture Notes in Computer Science, 13844; Macao, 04.12.–08.12.2022).
                        Cham 2022, pp.&#160;588–604. DOI: 10.1007/978-3-031-26316-3_35</bibl>
                    <bibl xml:id="choe_et_al_methods_2020">Junsuk Choe&#160;/ Seong Joon Oh&#160;/
                        Seungho Lee&#160;/ Sanghyuk Chun&#160;/ Zeynep Akata&#160;/ Hyunjung Shim:
                        Evaluating Weakly Supervised Object Localization Methods Right. In: 2020
                        IEEE&#160;/ CVF Conference on Computer Vision and Pattern Recognition (CVPR
                        2020; Seattle, 13.06.–19.06.2020). New York 2020, pp.&#160;3130–3139. DOI:
                        10.1109/CVPR42600.2020.00320</bibl>
                    <bibl xml:id="coy_bonsiepen_expertensystemtechnik_1989">Wolfgang Coy&#160;/ Lena
                        Bonsiepen: Erfahrung und Berechnung: Kritik der Expertensystemtechnik.
                        Berlin etc. 1989. <ptr type="gbv" cRef="272144045"/></bibl>
                    <bibl xml:id="gonthier_et_al_computervision_2018">Nicolas Gonthier&#160;/ Yann
                        Gousseau&#160;/ Said Ladjal&#160;/ Olivier Bonfait: Weakly Supervised Object
                        Detection in Artworks. In: Laura Leal-Taixé&#160;/ Stefan Roth (eds.):
                        Computer Vision&#160;– ECCV 2018 Workshops (=&#160;Lecture Notes in Computer
                        Science, 11130; Munich, 08.09.–14.09.2018). Cham 2018, pp.&#160;692–709.
                        DOI: 10.1007/978-3-030-11012-3_53</bibl>
                    <bibl xml:id="gupta_et_al_learning_2020">Tanmay Gupta&#160;/ Arash Vahdat&#160;/
                        Gal Chechik&#160;/ Xiaodong Yang&#160;/ Jan Kautz&#160;/ Derek Hoiem:
                        Contrastive Learning for Weakly Supervised Phrase Grounding. In: Andrea
                        Vedaldi&#160;/ Horst Bischof&#160;/ Thomas Brox&#160;/ Jan-Michael Frahm
                        (eds.): Computer Vision&#160;– ECCV 2020: 16<hi rend="super">th</hi>
                        European Conference (=&#160;Lecture Notes in Computer Science, 12348;
                        Glasgow, 23.08.–28.08.2020). Cham 2020, pp.&#160;752–768. DOI:
                        10.1007/978-3-030-58580-8_44</bibl>
                    <bibl xml:id="harmon_et_al_expertensysteme_1989">Paul Harmon&#160;/ Rex
                        Maus&#160;/ William Morrissey: Expertensysteme: Werkzeuge und Anwendungen.
                        München etc. 1989. <ptr type="gbv" cRef="025547569"/></bibl>
                    <bibl xml:id="hassija_et_al_black-box_2024">Vikas Hassija&#160;/ Vinay
                        Chamola&#160;/ Atmesh Mahapatra&#160;/ Abhinandan Singal&#160;/ Divyansh
                        Goel&#160;/ Kaizhu Huang&#160;/ Simone Scardapane&#160;/ Indro
                        Spinelli&#160;/ Mufti Mahmud&#160;/ Amir Hussain: Interpreting Black-Box
                        Models: A Review on Explainable Artificial Intelligence. In: Cognitive
                        Computation 16 (2024), no.&#160;1, pp.&#160;45–74. DOI: <ref target="https://doi.org/10.1007/s12559-023-10179-8">10.1007/s12559-023-10179-8</ref>
                    </bibl>
                    <bibl xml:id="impett_offert_arthistory_2022">Leonardo Impett&#160;/ Fabian
                        Offert: There Is a Digital Art History. In: Visual Resources 38 (2022),
                        no.&#160;2, pp.&#160;186–209. DOI: 10.1080/01973762.2024.2362466</bibl>
                    <bibl xml:id="jain_wallace_attention_2019">Sarthak Jain&#160;/ Byron C. Wallace:
                        Attention is not Explanation. In: Jill Burstein&#160;/ Christy Doran&#160;/
                        Thamar Solorio (eds.): Proceedings of the 2019 Conference of the North
                        American Chapter of the Association for Computational Linguistics: Human
                        Language Technologies (NAACL-HLT 2019). Minneapolis 2019,
                        pp.&#160;3543–3556. DOI: <ref target="https://doi.org/10.18653/v1/N19-1357">10.18653/v1/N19-1357</ref>
                    </bibl>
                    <bibl xml:id="jiang_et_al_maps_2021">Peng-Tao Jiang&#160;/ Chang-Bin
                        Zhang&#160;/ Qibin Hou&#160;/ Ming-Ming Cheng&#160;/ Yunchao Wei: LayerCAM:
                        Exploring Hierarchical Class Activation Maps for Localization. In: IEEE
                        Transactions on Image Processing 30 (2021), pp.&#160;5875–5888. DOI:
                        10.1109/TIP.2021.3089943</bibl>
                    <bibl xml:id="kazmierczak_et_al_clip_2023">Rémi Kazmierczak&#160;/ Eloïse
                        Berthier&#160;/ Goran Frehse&#160;/ Gianni Franchi: CLIP-QDA: An Explainable
                        Concept Bottleneck Model. In: Transactions on Machine Learning Research 05
                        (2024). arXiv. 30.11.2023. Version&#160;3: 31.05.2024. DOI: <ref target="https://arxiv.org/abs/2312.00110v3">abs/2312.00110v3</ref>
                    </bibl>
                    <bibl xml:id="li_et_al_blip_2023">Junnan Li&#160;/ Dongxu Li&#160;/ Silvio
                        Savarese&#160;/ Steven C. H. Hoi: BLIP-2: Bootstrapping Language-Image
                        Pre-training with Frozen Image Encoders and Large Language Models. In:
                        Andreas Krause&#160;/ Emma Brunskill&#160;/ Kyunghyun Cho&#160;/ Barbara
                        Engelhardt&#160;/ Sivan Sabato&#160;/ Jonathan Scarlett (eds.): Proceedings
                        of the 40<hi rend="super">th</hi> International Conference on Machine
                        Learning (Honolulu, US-HI, 23.07.–29.07.2023). PMLR 202 (2023),
                        pp.&#160;19730–19742. [<ref target="https://proceedings.mlr.press/v202/li23q.html">online</ref>]</bibl>
                    <bibl xml:id="li_et_al_explainability_2025">Yi Li&#160;/ Hualiang Wang&#160;/
                        Yiqun Duan&#160;/ Jiheng Zhang&#160;/ Xiaomeng Li: A Closer Look at the
                        Explainability of Contrastive Language-Image Pre-Training. In: Pattern
                        Recognition 162 (2025). DOI: 10.1016/j.patcog.2025.111409</bibl>
                    <bibl xml:id="milani_fraternali_dataset_2021">Federico Milani&#160;/ Piero
                        Fraternali: A Dataset and a Convolutional Model for Iconography
                        Classification in Paintings. In: ACM Journal on Computing and Cultural
                        Heritage 14 (2021), no.&#160;4, pp.&#160;1–18. DOI: 10.1145/3458885</bibl>
                    <bibl xml:id="miller_explanation_2019">Tim Miller: Explanation in Artificial
                        Intelligence: Insights from the Social Sciences. In: Artificial Intelligence
                        267 (2019), pp.&#160;1–38. DOI: 10.1016/j.artint.2018.07.007</bibl>
                    <bibl xml:id="offert_bell_imgs_2023">Fabian Offert&#160;/ Peter Bell: imgs.ai: A
                        Deep Visual Search Engine for Digital Art History. In: Anne Baillot&#160;/
                        Toma Tasovac&#160;/ Walter Scholger&#160;/ Georg Vogeler (eds.): DH 2023.
                        Collaboration as Opportunity. International Conference of the Alliance of
                        Digital Humanities Organizations (Graz, 10.07.–14.07.2023). Graz 2023. DOI: <ref target="https://doi.org/10.5281/zenodo.8107778">10.5281/zenodo.8107778</ref>
                    </bibl>
                    <bibl xml:id="oikarinen_weng_clip_2023">Tuomas Oikarinen&#160;/ Tsui-Wei Weng:
                        CLIP-Dissect: Automatic Description of Neuron Representations in Deep Vision
                        Networks. In: The 11<hi rend="super">th</hi> International Conference on
                        Learning Representations (Kigali, Rwanda, 01.05.–05.05.2023). 2023. [<ref target="https://openreview.net/pdf/a302e0072a6e15c8c0361c022bb9d3518f1a7127.pdf">online</ref>]</bibl>
                    <bibl xml:id="puppe_expertensystem_1988">Frank Puppe: Einführung in
                        Expertensysteme. Berlin etc. 1988. <ptr type="gbv" cRef="025185608"/></bibl>
                    <bibl xml:id="radford_et_al_models_2021">Alec Radford&#160;/ Jong Wook
                        Kim&#160;/ Chris Hallacy&#160;/ Aditya Ramesh&#160;/ Gabriel Goh&#160;/
                        Sandhini Agarwal&#160;/ Girish Sastry&#160;/ Amanda Askell&#160;/ Pamela
                        Mishkin&#160;/ Jack Clark&#160;/ Gretchen Krueger&#160;/ Ilya Sutskever:
                        Learning Transferable Visual Models From Natural Language Supervision. In:
                        Marina Meila&#160;/ Tong Zhang (eds.): Proceedings of the 38<hi rend="super"
                            >th</hi> International Conference on Machine Learning (Virtual,
                        18.07.–24.07.2021). PMLR 139 (2021), pp.&#160;8748–8763. [<ref target="http://proceedings.mlr.press/v139/radford21a.html">online</ref>]</bibl>
                    <bibl xml:id="reshetnikov_et_al_deart_2022">Artem Reshetnikov&#160;/
                        Maria-Cristina V. Marinescu&#160;/ Joaquim Moré López: DEArt: Dataset of
                        European Art. In: Leonid Karlinsky&#160;/ Tomer Michaeli &#160;/ Ko Nishino
                        (eds.): Computer Vision&#160;– ECCV 2022 Workshops (=&#160;Lecture Notes in
                        Computer Science, 13801; Tel Aviv, Israel, 23.10.–27.10.2022). Cham 2022,
                        pp.&#160;218–233. DOI: 10.1007/978-3-031-25056-9_15</bibl>
                    <bibl xml:id="saeed_omlin_xai_2023">Waddah Saeed&#160;/ Christian W. Omlin:
                        Explainable AI (XAI): A Systematic Meta-Survey of Current Challenges and
                        Future Opportunities. In: Knowledge-Based Systems 263 (2023). DOI:
                        10.1016/j.knosys.2023.110273</bibl>
                    <bibl xml:id="schneider_vollmer_poses_2024">Stefanie Schneider&#160;/ Ricarda
                        Vollmer: Poses of People in Art: A Dataset for Human Pose Estimation in
                        Digital Art History. In: ACM Journal on Computing and Cultural Heritage 17
                        (2024), no.&#160;4, pp.&#160;1–19. DOI: <ref target="https://doi.org/10.1145/3696455">10.1145/3696455</ref>
                    </bibl>
                    <bibl xml:id="schuhmann_et_al_laion_2021">Christoph Schuhmann&#160;/ Richard
                        Vencu&#160;/ Romain Beaumont&#160;/ Robert Kaczmarczyk&#160;/ Clayton
                        Mullis&#160;/ Aarush Katta&#160;/ Theo Coombes&#160;/ Jenia Jitsev&#160;/
                        Aran Komatsuzaki: LAION-400M: Open Dataset of CLIP-Filtered 400 Million
                        Image-Text Pairs. arXiv. 03.11.2021. DOI: <ref target="https://doi.org/10.48550/arXiv.2111.02114">10.48550/arXiv.2111.02114</ref>
                    </bibl>
                    <bibl xml:id="selvaraju_et_al_grad-cam_2017">Ramprasaath R. Selvaraju&#160;/
                        Michael Cogswell&#160;/ Abhishek Das&#160;/ Ramakrishna Vedantam&#160;/ Devi
                        Parikh&#160;/ Dhruv Batra: Grad-CAM: Visual Explanations from Deep Networks
                        via Gradient-Based Localization. In: 2017 IEEE International Conference on
                        Computer Vision (ICCV 2017; Venice, 22.10.–29.10.2017). New York 2017,
                        pp.&#160;618–626. DOI: 10.1109/ICCV.2017.74</bibl>
                    <bibl xml:id="speith_review_2022">Timo Speith: A Review of Taxonomies of
                        Explainable Artificial Intelligence (XAI) Methods. In: FAccT ’22:
                        Proceedings of the 2022 ACM Conference on Fairness, Accountability, and
                        Transparency (Seoul, 21.06.–24.06.2022). New York 2022, pp.&#160;2239–2250.
                        DOI: 10.1145/3531146.3534639</bibl>
                    <bibl xml:id="springstein_et_al_iART_2021">Matthias Springstein&#160;/ Stefanie
                        Schneider&#160;/ Javad Rahnama&#160;/ Eyke Hüllermeier&#160;/ Hubertus
                        Kohle&#160;/ Ralph Ewerth: iART: A Search Engine for Art-Historical Images
                        to Support Research in the Humanities. In: Heng Tao Shen&#160;/ Yueting
                        Zhuang&#160;/ John R. Smith&#160;/ Yang Yang&#160;/ Pablo César&#160;/
                        Florian Metze&#160;/ Balakrishnan Prabhakaran (eds.): MM ’21: Proceedings of
                        the 29<hi rend="super">th</hi> ACM International Conference on Multimedia
                        (Virtual, 20.10.–24.10.2021). New York 2021, pp.&#160;2801–2803. DOI: 10.1145/3474085.3478564</bibl>
                    <bibl xml:id="wang_et_al_score-CAM_2020">Haofan Wang&#160;/ Zifan Wang&#160;/
                        Mengnan Du&#160;/ Fan Yang&#160;/ Zijian Zhang &#160;/ Sirui Ding&#160;/
                        Piotr Mardziel&#160;/ Xia Hu: Score-CAM: Score-Weighted Visual Explanations
                        for Convolutional Neural Networks. In: 2020 IEEE&#160;/ CVF Conference on
                        Computer Vision and Pattern Recognition (CVPR Workshops 2020). Seattle 2020,
                        pp.&#160;111–119. DOI: 10.1109/CVPRW50498.2020.00020</bibl>
                    <bibl xml:id="zhai_et_al_transfer_2022">Xiaohua Zhai&#160;/ Xiao Wang&#160;/
                        Basil Mustafa&#160;/ Andreas Steiner&#160;/ Daniel Keysers&#160;/ Alexander
                        Kolesnikov&#160;/ Lucas Beyer: LiT: Zero-Shot Transfer with Locked-Image
                        Text Tuning. In: IEEE&#160;/ CVF Conference on Computer Vision and Pattern
                        Recognition (CVPR 2022). New Orleans 2022, pp.&#160;18102–18112. DOI: 10.1109/CVPR52688.2022.01759</bibl>
                </listBibl>
            </div>
        </back>
    </text>
</TEI>
