{
  "schema_version": 2,
  "relationship_export_date": "2026-09-21",
  "relationships": {
    "mentions": [
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "Method"
        ],
        "name": "Channel State Information",
        "slug": "channel-state-information",
        "definition": null,
        "description": "Channel State Information (CSI) is the per-subcarrier complex channel response sampled by an OFDM receiver during normal Wi-Fi packet reception. Each CSI sample is a vector of complex numbers (one per subcarrier per TX/RX antenna pair) describing how the multipath channel attenuates and phase-shifts each frequency bin, which makes it a far richer sensing observable than scalar RSSI. CSI is the foundational primitive of WiFi sensing — every downstream task in this thesis (movement detection, occupancy estimation, crowd dynamics) starts from a CSI time-series.",
        "aliases": [
          "Channel State Information",
          "Channel State Information (CSI)",
          "CSI (Channel State Information)",
          "CSI",
          "CSI measurement",
          "CSI measurements"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "Hardware"
        ],
        "name": "CSI Monitor Station",
        "slug": "csi-monitor-station",
        "definition": null,
        "description": "A CSI monitor station is a host (typically Raspberry Pi, OpenWrt router, or Linux laptop) running a CSI-extraction NIC in **monitor mode**: it observes Wi-Fi traffic between other parties without participating, and logs per-frame CSI for sensing analysis. This is the \"passive sniffer\" deployment pattern that the thesis architecture relies on — fixed monitor stations placed at known coordinates around the monitored area continuously log CSI from any cooperating Wi-Fi traffic in the room. Distinct from an [[access-point]] because it never serves clients; only listens.",
        "aliases": [
          "CSI Monitor Station",
          "CSI monitor",
          "CSI sniffer",
          "WiFi monitor station",
          "Wi-Fi monitor station",
          "CSI receiver",
          "monitor mode receiver"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "Problem"
        ],
        "name": "Indoor Localization",
        "slug": "indoor-localization",
        "definition": null,
        "description": "Estimating the position of a person or device inside a building, where GNSS is unavailable or unreliable. The problem comes in two flavors that share most of the literature: *device-based* (a phone or tag reports RSSI/CSI/UWB readings used to recover its own coordinates) and *device-free* (the environment is monitored and bodies are localized by the perturbation they cause to a wireless link). For the thesis, BLE provides device-based ground-truth trajectories during calibration campaigns; CSI handles device-free continuous monitoring between them.",
        "aliases": [
          "Indoor Localization",
          "Indoor localisation",
          "Indoor Positioning",
          "indoor location estimation",
          "in-building localization",
          "device-free indoor localization",
          "positioning",
          "localization"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "Dataset"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": null,
        "description": "OPERAnet is a **synchronised multimodal human-sensing dataset** from the University of Bristol OPERA project (Bocus et al., *Scientific Data* 2022). It records the **same activities simultaneously through four modalities** — WiFi CSI, Passive WiFi Radar (PWR), Ultra-Wideband (UWB, two passive receivers), and Microsoft Kinect v2 skeletons — so CSI-only, radar-only, UWB-only and vision pipelines can be compared on identical ground truth. ~8 hours of annotated measurements, up to 6 participants, two furnished office rooms. Intended for **human activity recognition (HAR), device-free/non-collaborative localization, and crowd counting**.",
        "aliases": [
          "OPERAnet",
          "OPERAnet dataset",
          "OPERA-net",
          "OPERA net"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "DictionaryEntry"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": "OPERAnet is a publicly available CSI-based Wi-Fi sensing dataset designed to support research in indoor human activity recognition and localization, collected using commodity Wi-Fi hardware across multiple environments and subjects. It matters for the field because it provides a standardized benchmark that facilitates reproducibility and cross-study comparison of sensing models, addressing a critical gap in a research area where inconsistent data collection practices have historically hindered scientific validation. The dataset is notable for its multi-environment and multi-subject scope, though specific variants or versioned releases may differ in the number of antennas, subcarriers, or activity classes covered depending on the configuration used in a given experimental setup.",
        "description": "OPERAnet is a publicly available CSI-based Wi-Fi sensing dataset designed to support research in indoor human activity recognition and localization, collected using commodity Wi-Fi hardware across multiple environments and subjects. It matters for the field because it provides a standardized benchmark that facilitates reproducibility and cross-study comparison of sensing models, addressing a critical gap in a research area where inconsistent data collection practices have historically hindered scientific validation. The dataset is notable for its multi-environment and multi-subject scope, though specific variants or versioned releases may differ in the number of antennas, subcarriers, or activity classes covered depending on the configuration used in a given experimental setup.",
        "aliases": [
          "OPERAnet"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_04944cb19a12",
        "labels": [
          "Keyword"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": "OPERAnet is a large-scale, multi-environment CSI dataset collected using commodity IEEE 802.11ac hardware that serves as a benchmark for training and evaluating machine learning models for tasks such as indoor localization, activity recognition, and gesture detection by providing labeled channel state information measurements across diverse real-world propagation conditions.",
        "description": "OPERAnet is a large-scale, multi-environment CSI dataset collected using commodity IEEE 802.11ac hardware that serves as a benchmark for training and evaluating machine learning models for tasks such as indoor localization, activity recognition, and gesture detection by providing labeled channel state information measurements across diverse real-world propagation conditions.",
        "aliases": [
          "OPERAnet"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_069388bcbc50",
        "labels": [
          "Dataset"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": null,
        "description": "OPERAnet is a **synchronised multimodal human-sensing dataset** from the University of Bristol OPERA project (Bocus et al., *Scientific Data* 2022). It records the **same activities simultaneously through four modalities** — WiFi CSI, Passive WiFi Radar (PWR), Ultra-Wideband (UWB, two passive receivers), and Microsoft Kinect v2 skeletons — so CSI-only, radar-only, UWB-only and vision pipelines can be compared on identical ground truth. ~8 hours of annotated measurements, up to 6 participants, two furnished office rooms. Intended for **human activity recognition (HAR), device-free/non-collaborative localization, and crowd counting**.",
        "aliases": [
          "OPERAnet",
          "OPERAnet dataset",
          "OPERA-net",
          "OPERA net"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_069388bcbc50",
        "labels": [
          "DictionaryEntry"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": "OPERAnet is a publicly available CSI-based Wi-Fi sensing dataset designed to support research in indoor human activity recognition and localization, collected using commodity Wi-Fi hardware across multiple environments and subjects. It matters for the field because it provides a standardized benchmark that facilitates reproducibility and cross-study comparison of sensing models, addressing a critical gap in a research area where inconsistent data collection practices have historically hindered scientific validation. The dataset is notable for its multi-environment and multi-subject scope, though specific variants or versioned releases may differ in the number of antennas, subcarriers, or activity classes covered depending on the configuration used in a given experimental setup.",
        "description": "OPERAnet is a publicly available CSI-based Wi-Fi sensing dataset designed to support research in indoor human activity recognition and localization, collected using commodity Wi-Fi hardware across multiple environments and subjects. It matters for the field because it provides a standardized benchmark that facilitates reproducibility and cross-study comparison of sensing models, addressing a critical gap in a research area where inconsistent data collection practices have historically hindered scientific validation. The dataset is notable for its multi-environment and multi-subject scope, though specific variants or versioned releases may differ in the number of antennas, subcarriers, or activity classes covered depending on the configuration used in a given experimental setup.",
        "aliases": [
          "OPERAnet"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_069388bcbc50",
        "labels": [
          "Keyword"
        ],
        "name": "OPERAnet",
        "slug": "operanet",
        "definition": "OPERAnet is a large-scale, multi-environment CSI dataset collected using commodity IEEE 802.11ac hardware that serves as a benchmark for training and evaluating machine learning models for tasks such as indoor localization, activity recognition, and gesture detection by providing labeled channel state information measurements across diverse real-world propagation conditions.",
        "description": "OPERAnet is a large-scale, multi-environment CSI dataset collected using commodity IEEE 802.11ac hardware that serves as a benchmark for training and evaluating machine learning models for tasks such as indoor localization, activity recognition, and gesture detection by providing labeled channel state information measurements across diverse real-world propagation conditions.",
        "aliases": [
          "OPERAnet"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_4ea99acc4119",
        "labels": [
          "Hardware"
        ],
        "name": "IEEE 802.11",
        "slug": "ieee-802-11",
        "definition": null,
        "description": "IEEE 802.11 is the family of standards defining Wireless LAN PHY and MAC layers — the substrate of every CSI sensing paper in the bibliography. This umbrella entry canonicalises the bare aliases (`WiFi`, `Wi-Fi`, `WLAN`); the per-generation notes ([[ieee-802-11n]], [[ieee-802-11ac]], [[ieee-802-11ax]], [[ieee-802-11bf]]) capture the CSI-relevant differences between generations.",
        "aliases": [
          "IEEE 802.11",
          "802.11",
          "WiFi",
          "Wi-Fi",
          "WLAN",
          "wireless LAN",
          "WiFi standard",
          "Wi-Fi standard"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "bocus2022_ce7f_claim_4ea99acc4119",
        "labels": [
          "Hardware"
        ],
        "name": "Ultra-Wideband",
        "slug": "uwb",
        "definition": null,
        "description": "Ultra-Wideband (UWB, IEEE 802.15.4z) is a short-range radio with extremely wide instantaneous bandwidth (>500 MHz), which gives it the **best ranging accuracy of any consumer wireless** — typically 10-30 cm in indoor environments via Two-Way Ranging or TDoA. UWB is the strongest competing modality to BLE for indoor localization: better accuracy, similar power, but more expensive infrastructure. Apple's U1 / U2 chip (iPhone 11+, Watch 6+) and Qorvo's DW3000 are bringing UWB to mass-market scale. For the thesis, UWB is the high-accuracy comparison baseline against which BLE-calibration accuracy is benchmarked, and a candidate for ground-truth trajectory measurement during calibration campaigns.",
        "aliases": [
          "Ultra-Wideband",
          "UWB",
          "ultra wideband",
          "DW1000",
          "DW3000",
          "DWM1001",
          "Decawave",
          "Apple U1",
          "U1 chip"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "ingram2025_642d_claim_0631056d69b7",
        "labels": [
          "DictionaryEntry"
        ],
        "name": "Logistic Regression",
        "slug": "logistic-regression",
        "definition": "Logistic Regression is a supervised statistical classification algorithm that models the probability of a discrete class label by applying a logistic (sigmoid) function to a linear combination of input features, making it well-suited for binary or multiclass categorization tasks such as distinguishing activity types, occupancy states, or spatial sections from CSI amplitude or phase features. In WiFi sensing research, it serves as a lightweight, interpretable baseline classifier that requires minimal computational resources, enabling deployment on constrained hardware like ESP32 modules and providing a reproducible performance benchmark against which more complex models such as neural networks can be fairly compared. Key variants include One-vs-Rest (OvR) and multinomial logistic regression for handling multiclass problems, as well as regularized forms (L1/Lasso and L2/Ridge penalties) that mitigate overfitting when the number of CSI subcarrier features is large relative to the number of labeled training samples.",
        "description": "Logistic Regression is a supervised statistical classification algorithm that models the probability of a discrete class label by applying a logistic (sigmoid) function to a linear combination of input features, making it well-suited for binary or multiclass categorization tasks such as distinguishing activity types, occupancy states, or spatial sections from CSI amplitude or phase features. In WiFi sensing research, it serves as a lightweight, interpretable baseline classifier that requires minimal computational resources, enabling deployment on constrained hardware like ESP32 modules and providing a reproducible performance benchmark against which more complex models such as neural networks can be fairly compared. Key variants include One-vs-Rest (OvR) and multinomial logistic regression for handling multiclass problems, as well as regularized forms (L1/Lasso and L2/Ridge penalties) that mitigate overfitting when the number of CSI subcarrier features is large relative to the number of labeled training samples.",
        "aliases": [
          "Logistic Regression"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "ingram2025_642d_claim_0631056d69b7",
        "labels": [
          "Keyword"
        ],
        "name": "Logistic Regression",
        "slug": "logistic-regression",
        "definition": "Logistic Regression is used in WiFi/CSI sensing research as a linear classification model that maps extracted CSI amplitude or phase features—such as mean, variance, or PCA-reduced components—to discrete activity or gesture classes by learning a set of weight coefficients optimized via maximum likelihood estimation.",
        "description": "Logistic Regression is used in WiFi/CSI sensing research as a linear classification model that maps extracted CSI amplitude or phase features—such as mean, variance, or PCA-reduced components—to discrete activity or gesture classes by learning a set of weight coefficients optimized via maximum likelihood estimation.",
        "aliases": [
          "Logistic Regression"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "ingram2025_642d_claim_0631056d69b7",
        "labels": [
          "Method"
        ],
        "name": "Logistic Regression",
        "slug": "logistic-regression",
        "definition": null,
        "description": "Logistic regression is a generalised linear model that maps features to a class probability via the logistic / softmax link. It is the calibration-friendly linear baseline in CSI sensing — interpretable coefficients, well-defined uncertainty, fast inference — and is the default head on top of CNN/Transformer feature backbones.",
        "aliases": [
          "Logistic Regression",
          "logit model",
          "softmax regression"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "tran2024_104b_claim_6f0803d4003b",
        "labels": [
          "TopicNote"
        ],
        "name": "Evaluation",
        "slug": "thesis-evaluation",
        "definition": null,
        "description": "Evaluation in WiFi CSI-based sensing research encompasses the systematic assessment of system accuracy, generalizability, scalability, and real-world deployability across tasks such as occupancy estimation, crowd counting, activity recognition, and multi-user localization. Studies employ a range of methodological approaches including benchmark dataset construction and standardized comparison frameworks, deep learning model evaluation pipelines, multi-transceiver experimental setups, and formal modeling techniques such as SysML and Petri Nets to assess network and system performance under controlled and naturalistic conditions. Evaluation metrics commonly span counting accuracy, classification precision, latency, energy efficiency, and alert delivery reliability, with platforms like SenseFi and WiMANS providing structured benchmarks to enable reproducible cross-study comparisons. A prominent open challenge is environment dependence, whereby models trained in one physical setting degrade significantly when deployed in another, raising concerns about transferability and robustness that remain only partially addressed through data augmentation or domain adaptation. Emerging trends include end-to-end evaluation frameworks that integrate sensing, inference, and communication layers, energy-aware sensing evaluation leveraging ambient traffic, and the growing emphasis on multi-zone, multi-user, and real-world deployment scenarios as the field moves beyond controlled laboratory benchmarks toward practical validation.",
        "aliases": [
          "Evaluation"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "tran2024_104b_claim_a0d3a32d50de",
        "labels": [
          "Hardware"
        ],
        "name": "Access Point",
        "slug": "access-point",
        "definition": null,
        "description": "An 802.11 Access Point (AP) is the infrastructure-side endpoint of a Wi-Fi link — typically a wall-mounted or ceiling-mounted device combining a Wi-Fi radio, an Ethernet uplink, and routing functions. For CSI sensing the AP plays one of two roles: **transmitter of probe / data frames** that another monitor station decodes, or **monitor itself** if firmware-patched. Common research APs include TP-Link N750/AC1750 (Atheros, OpenWrt-friendly), ASUS RT-AC86U (Broadcom, Nexmon-friendly), and any AP based on the [[atheros-csi-tool]] or [[nexmon-csi]] supported chipsets. The thesis deployment uses APs as the fixed CSI vantage points that observe the monitored area.",
        "aliases": [
          "Access Point",
          "Access Points",
          "AP",
          "APs",
          "Wi-Fi access point",
          "WiFi access point",
          "Wi-Fi access points",
          "router",
          "routers",
          "WiFi router",
          "Wi-Fi router",
          "wireless router"
        ],
        "provenance": "curated",
        "auto_generated": false,
        "evidence": {}
      },
      {
        "claim": "zhang2024_aea3_claim_23d4516e921c",
        "labels": [
          "DictionaryEntry"
        ],
        "name": "knowledge distillation",
        "slug": "knowledge-distillation",
        "definition": "Knowledge distillation is a model compression and transfer technique in which a smaller \"student\" network is trained to mimic the output distributions, intermediate representations, or behavioral patterns of a larger, more capable \"teacher\" network, rather than being trained solely on hard ground-truth labels. In WiFi CSI-based human sensing, it matters because it enables the deployment of lightweight models on resource-constrained edge devices while preserving much of the predictive accuracy achieved by complex deep learning architectures, and it also facilitates cross-domain adaptation by transferring learned feature representations from data-rich source domains to data-scarce target domains. Key variants relevant to this field include response-based distillation, which matches final output logits or probability distributions, feature-based distillation, which aligns intermediate layer activations, and self-distillation or few-shot distillation schemes that are particularly useful in scenarios like DASECount where labeled target-domain samples are extremely limited.",
        "description": "Knowledge distillation is a model compression and transfer technique in which a smaller \"student\" network is trained to mimic the output distributions, intermediate representations, or behavioral patterns of a larger, more capable \"teacher\" network, rather than being trained solely on hard ground-truth labels. In WiFi CSI-based human sensing, it matters because it enables the deployment of lightweight models on resource-constrained edge devices while preserving much of the predictive accuracy achieved by complex deep learning architectures, and it also facilitates cross-domain adaptation by transferring learned feature representations from data-rich source domains to data-scarce target domains. Key variants relevant to this field include response-based distillation, which matches final output logits or probability distributions, feature-based distillation, which aligns intermediate layer activations, and self-distillation or few-shot distillation schemes that are particularly useful in scenarios like DASECount where labeled target-domain samples are extremely limited.",
        "aliases": [
          "knowledge distillation"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "zhang2024_aea3_claim_23d4516e921c",
        "labels": [
          "Keyword"
        ],
        "name": "knowledge distillation",
        "slug": "knowledge-distillation",
        "definition": "In WiFi/CSI sensing research, knowledge distillation is used to transfer the learned channel state information feature representations from a large, computationally expensive teacher network to a lightweight student network, enabling accurate gesture recognition, human activity classification, or indoor localization on resource-constrained edge devices while preserving model performance.",
        "description": "In WiFi/CSI sensing research, knowledge distillation is used to transfer the learned channel state information feature representations from a large, computationally expensive teacher network to a lightweight student network, enabling accurate gesture recognition, human activity classification, or indoor localization on resource-constrained edge devices while preserving model performance.",
        "aliases": [
          "knowledge distillation"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      },
      {
        "claim": "zhang2024_aea3_claim_23d4516e921c",
        "labels": [
          "Method"
        ],
        "name": "knowledge distillation",
        "slug": "knowledge-distillation",
        "definition": null,
        "description": "Knowledge distillation is a model compression and transfer technique in which a smaller \"student\" network is trained to mimic the output distributions, intermediate representations, or behavioral patterns of a larger, more capable \"teacher\" network, rather than being trained solely on hard ground-truth labels. In WiFi CSI-based human sensing, it matters because it enables the deployment of lightweight models on resource-constrained edge devices while preserving much of the predictive accuracy achieved by complex deep learning architectures, and it also facilitates cross-domain adaptation by transferring learned feature representations from data-rich source domains to data-scarce target domains. Key variants relevant to this field include response-based distillation, which matches final output logits or probability distributions, feature-based distillation, which aligns intermediate layer activations, and self-distillation or few-shot distillation schemes that are particularly useful in scenarios like DASECount where labeled target-domain samples are extremely limited.",
        "aliases": [
          "knowledge distillation"
        ],
        "provenance": "curated",
        "auto_generated": true,
        "evidence": {}
      }
    ],
    "groups": [
      {
        "paper": "bocus2022_ce7f",
        "tags": [],
        "topics": [
          {
            "auto_generated": false,
            "name": "Smart Environments",
            "provenance": "curated",
            "slug": "domain-smart-environments"
          },
          {
            "auto_generated": false,
            "name": "Csi Sensing",
            "provenance": "curated",
            "slug": "thesis-csi-sensing"
          },
          {
            "auto_generated": true,
            "name": "Experimental",
            "provenance": "curated",
            "slug": "method-experimental"
          },
          {
            "auto_generated": true,
            "name": "Wireless Sensing",
            "provenance": "curated",
            "slug": "domain-wireless-sensing"
          }
        ],
        "communities": [
          {
            "name": "WiFi CSI Human Sensing",
            "id": 61,
            "paper_count": 92
          }
        ]
      },
      {
        "paper": "ingram2025_642d",
        "tags": [
          {
            "name": "Iot Systems",
            "tag": "domain/iot-systems"
          },
          {
            "name": "Machine Learning",
            "tag": "method/machine-learning"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Iot Systems",
            "provenance": "curated",
            "slug": "domain-iot-systems"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Machine Learning",
            "provenance": "curated",
            "slug": "method-machine-learning"
          }
        ],
        "communities": []
      },
      {
        "paper": "janssens2024_2f53",
        "tags": [],
        "topics": [
          {
            "auto_generated": false,
            "name": "Crowd Modeling",
            "provenance": "curated",
            "slug": "thesis-crowd-modeling"
          },
          {
            "auto_generated": false,
            "name": "Csi Sensing",
            "provenance": "curated",
            "slug": "thesis-csi-sensing"
          },
          {
            "auto_generated": true,
            "name": "Machine Learning",
            "provenance": "curated",
            "slug": "method-machine-learning"
          },
          {
            "auto_generated": true,
            "name": "Experimental",
            "provenance": "curated",
            "slug": "method-experimental"
          },
          {
            "auto_generated": false,
            "name": "Wsn",
            "provenance": "curated",
            "slug": "domain-wsn"
          },
          {
            "auto_generated": true,
            "name": "Wireless Sensing",
            "provenance": "curated",
            "slug": "domain-wireless-sensing"
          }
        ],
        "communities": [
          {
            "name": "WiFi CSI Occupancy Detection",
            "id": 106,
            "paper_count": 18
          }
        ]
      },
      {
        "paper": "kroll2024_3ce7",
        "tags": [
          {
            "name": "Iot Systems",
            "tag": "domain/iot-systems"
          },
          {
            "name": "Deep Learning",
            "tag": "method/deep-learning"
          },
          {
            "name": "Machine Learning",
            "tag": "method/machine-learning"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Iot Systems",
            "provenance": "curated",
            "slug": "domain-iot-systems"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Machine Learning",
            "provenance": "curated",
            "slug": "method-machine-learning"
          },
          {
            "auto_generated": true,
            "name": "Deep Learning",
            "provenance": "curated",
            "slug": "method-deep-learning"
          }
        ],
        "communities": [
          {
            "name": "Scientific Literature Information Extraction",
            "id": 151,
            "paper_count": 22
          }
        ]
      },
      {
        "paper": "tai2024_5a71",
        "tags": [
          {
            "name": "Iot Systems",
            "tag": "domain/iot-systems"
          },
          {
            "name": "Survey",
            "tag": "method/survey"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Iot Systems",
            "provenance": "curated",
            "slug": "domain-iot-systems"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Survey",
            "provenance": "curated",
            "slug": "method-survey"
          }
        ],
        "communities": [
          {
            "name": "AI Integration in Library Systems",
            "id": 200,
            "paper_count": 4
          }
        ]
      },
      {
        "paper": "tran2024_104b",
        "tags": [
          {
            "name": "Smart Environments",
            "tag": "domain/smart-environments"
          },
          {
            "name": "Deep Learning",
            "tag": "method/deep-learning"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Iot Systems",
            "provenance": "curated",
            "slug": "domain-iot-systems"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Deep Learning",
            "provenance": "curated",
            "slug": "method-deep-learning"
          }
        ],
        "communities": [
          {
            "name": "RAG LLMs Scientific Literature Analysis",
            "id": 0,
            "paper_count": 22
          }
        ]
      },
      {
        "paper": "yamamoto2025_1935",
        "tags": [
          {
            "name": "Smart Environments",
            "tag": "domain/smart-environments"
          },
          {
            "name": "Deep Learning",
            "tag": "method/deep-learning"
          },
          {
            "name": "Experimental",
            "tag": "method/experimental"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Smart Environments",
            "provenance": "curated",
            "slug": "domain-smart-environments"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Experimental",
            "provenance": "curated",
            "slug": "method-experimental"
          },
          {
            "auto_generated": true,
            "name": "Deep Learning",
            "provenance": "curated",
            "slug": "method-deep-learning"
          }
        ],
        "communities": [
          {
            "name": "RAG LLMs Scientific Literature Analysis",
            "id": 0,
            "paper_count": 22
          }
        ]
      },
      {
        "paper": "zhang2024_aea3",
        "tags": [
          {
            "name": "Iot Systems",
            "tag": "domain/iot-systems"
          },
          {
            "name": "Machine Learning",
            "tag": "method/machine-learning"
          },
          {
            "name": "Supporting",
            "tag": "thesis/supporting"
          }
        ],
        "topics": [
          {
            "auto_generated": false,
            "name": "Iot Systems",
            "provenance": "curated",
            "slug": "domain-iot-systems"
          },
          {
            "auto_generated": true,
            "name": "Supporting",
            "provenance": "curated",
            "slug": "thesis-supporting"
          },
          {
            "auto_generated": true,
            "name": "Machine Learning",
            "provenance": "curated",
            "slug": "method-machine-learning"
          }
        ],
        "communities": [
          {
            "name": "RAG LLMs Scientific Literature Analysis",
            "id": 0,
            "paper_count": 22
          }
        ]
      }
    ]
  },
  "id": "jcdl-2026-09-21",
  "captured_at": "2026-09-20T23:26:18Z",
  "source_revision": "d7066fe7290aeb4375ba23c53df4bf15e8998728",
  "title": "Remembering the Sentence",
  "paper_doi": "10.1145/3805696.3846030",
  "selection": "Purposive selection from eight vault papers: digital libraries, retrieval and wireless sensing. Includes four previously audited examples and one paper-level reading result. This collection is not a random sample or an evaluation set. Text and identifiers are preserved from the graph; topic labels and tour explanations are editorial.",
  "census": {
    "vector_papers": 813,
    "vector_chunks": 84241,
    "graph_vault_papers": 840,
    "total_claims": 82371,
    "paragraph_claims": 80247,
    "paper_claims": 2124
  },
  "audit_source": "publications/jcdl-claim-graph/data/anchor-spotcheck-2026-07-03.md",
  "audit_method": "Historical systematic sample of 100 from 6,244 text-retaining claims, every 62nd in identifier order. Agent-proposed verdicts, researcher ratification; not independent blind annotation. This demo includes selected examples, not a new audit.",
  "papers": [
    {
      "key": "bocus2022_ce7f",
      "title": "OPERAnet, a multimodal activity recognition dataset acquired from radio frequency and vision-based sensors",
      "year": 2022,
      "doi": "10.1038/s41597-022-01573-2",
      "vault_path": "literature/OPERAnet, a multimodal activity recognition dataset acquired from radio frequency and vision-based sensors.md",
      "topic": "Wireless sensing"
    },
    {
      "key": "ingram2025_642d",
      "title": "Learning from LLM Disagreement in Retrieval Evaluation",
      "year": 2025,
      "doi": "10.1109/jcdl67857.2025.00024",
      "vault_path": "literature/Learning from LLM Disagreement in Retrieval Evaluation.md",
      "topic": "Search & retrieval"
    },
    {
      "key": "janssens2024_2f53",
      "title": "Device-Free Crowd Size Estimation Using Wireless Sensing on Subway Platforms",
      "year": 2024,
      "doi": "10.3390/app14209386",
      "vault_path": "literature/Device-Free Crowd Size Estimation Using Wireless Sensing on Subway Platforms.md",
      "topic": "Wireless sensing"
    },
    {
      "key": "kroll2024_3ce7",
      "title": "A Library Perspective on Supervised Text Processing in Digital Libraries: An Investigation in the Biomedical Domain",
      "year": 2024,
      "doi": "10.1145/3677389.3702557",
      "vault_path": "literature/A Library Perspective on Supervised Text Processing in Digital Libraries An Investigation in the Biomedical Domain.md",
      "topic": "Digital libraries"
    },
    {
      "key": "tai2024_5a71",
      "title": "Integrating AI into Library Systems: A Perspective on Applications and Challenges",
      "year": 2024,
      "doi": "10.1145/3677389.3702568",
      "vault_path": "literature/Integrating AI into Library Systems A Perspective on Applications and Challenges.md",
      "topic": "Digital libraries"
    },
    {
      "key": "tran2024_104b",
      "title": "Retrieval Augmented Generation for Historical Newspapers",
      "year": 2024,
      "doi": "10.1145/3677389.3702542",
      "vault_path": "literature/Retrieval Augmented Generation for Historical Newspapers.md",
      "topic": "Search & retrieval"
    },
    {
      "key": "yamamoto2025_1935",
      "title": "Scaffolding Inquiry-Oriented Web Search using LLM-based Question Generation",
      "year": 2025,
      "doi": "10.1109/jcdl67857.2025.00021",
      "vault_path": "literature/Scaffolding Inquiry-Oriented Web Search using LLM-based Question Generation.md",
      "topic": "Search & retrieval"
    },
    {
      "key": "zhang2024_aea3",
      "title": "Exploring Efficient Optimization Techniques in Online Retrieval-Augmented Generation Application",
      "year": 2024,
      "doi": "10.1145/3677389.3702522",
      "vault_path": "literature/Exploring Efficient Optimization Techniques in Online Retrieval-Augmented Generation Application.md",
      "topic": "Search & retrieval"
    }
  ],
  "claims": [
    {
      "paper_key": "tran2024_104b",
      "id": "tran2024_104b_claim_5011d54ce58b",
      "text": "Dense retrieval produces superior results compared to sparse retrieval (BM25) in historical newspaper retrieval tasks due to its ability to capture semantic and contextual meaning.",
      "type": "finding",
      "paragraph_id": "tran2024_104b_5.2",
      "paragraph": "where N is the number of features in the vector representation, P is all the retrieved paragraphs, r is the answer and 𝑝 𝑖 is the value of the vector representation of 𝑝 at index 𝑖.   1  shows all the results from the different methods: sparse retrieval (BM25), dense retrieval with cosine similarity (E5), the combination of dense retrieval with a cross-encoder model (E5+Cohere), and the reranker after injecting information (E5+Cohere+NER). As expected, dense retrieval produces a superior result compared to sparse retrieval thanks to its ability to capture the semantic and context meaning of a sentence. On the other hand, BM25 is highly dependent on vocabulary matching. In addition, we can see that with the normal reranking method (E5+Cohere), the system can increase its performance by nearly 0.1 points for NDCG. However, injecting NER seems to worsen the final result. This could be due in large part to the fact that using NER could cause misleading information in certain situations where named entities do not directly contribute to the key features of a sentence. Another factor might come from the errors of the NER extraction model itself. Regarding the results of the answer generation module with E5 retrieval shown in Table  2 , in the three languages, Finnish shows the lowest results among the three languages. This might show the inability of LLaMA3 to handle some languages with fewer resources. Meanwhile, the difference created by the BERTscore between English and French is significantly higher than that of other metrics where the gaps are unnoticeable. We computed the linear correlations between these scores to analyze how well each metric relates to each other across languages.   All three languages share a similar result, where the BERTscore and cosine similarity show the highest correlation with each other, greater than 0.7 between each pair. The reason behind this might involve the fact that both scores work on dense embedding vectors, making them similar. In addition to this, the quality index is slightly more correlated with other scores compared to the LLM score in the English data (Table  3 ). However, the reverse trend can be observed for French and Finnish, in Tables  4  and  5 , respectively. The quality index can be seen to have a very low correlation in these languages. In contrast, the LLM score has a much higher correlation, with roughly 0.8 in relation to the BERTscore and cosine similarity in the French data.",
      "section": "Quantitative Evaluation",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 3,
      "stored_claim_page": 1,
      "stored_claim_section": "Quantitative Evaluation",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "tran2024_104b",
      "id": "tran2024_104b_claim_6f0803d4003b",
      "text": "The evaluation used the French and Finnish subsets of the Miracl dataset, which is a multilingual dataset for information retrieval evaluation.",
      "type": "methodology",
      "paragraph_id": "tran2024_104b_5.0",
      "paragraph": "4.1.1 Setup. For this evaluation, we used the French and Finnish subsets of Miracl  [20] , a multilingual dataset used for the evaluation of information retrieval. In addition to French and Finnish, we added English to evaluate the answer generation module. Since this dataset is already split into passages, we directly indexed each chunk in the database. In addition, Miracl provides a title as metadata for each of these passages, which we indexed and used for title retrieval. Lastly, we set the temperature of LLaMA3 to 0.3 for the final answer generation module.",
      "section": "Quantitative Evaluation",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 3,
      "stored_claim_page": 1,
      "stored_claim_section": "Quantitative Evaluation",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "tran2024_104b",
      "id": "tran2024_104b_claim_a0d3a32d50de",
      "text": "The RAG system comprises four main components: a query router, a base retrieval model, a rerank module, and an LLM prompt aggregator.",
      "type": "methodology",
      "paragraph_id": "tran2024_104b_3.0",
      "paragraph": "Four our RAG system pipeline we first create a database using the small multilingual model E5  4  to embed all documents into vector representations. We also add metadata to the text during this step, such as the article's title. We then index the E5 embeddings in the open-source vector database Chroma foot_4  . The document's similarity calculation is calculated using cosine similarity and the retrieval process will employ maximal marginal relevance (MMR)  [2] . We use the same procedure on the title or summary of each article to create a second database which is in the role of a semantic router in our pipeline. After the creation of the necessary databases, the main system make use of these and comprises of four main components: a query router where we redirect the behavior of the system based on the user's question, a base retrieval model where some of the best documents related to the question are retrieved and they will be ranked using a rerank module to filter out some irrelevant ones. And lastly, all of these retrieved documents are aggregated and forward to the prompt of an LLM to generate the final answer given a user's query. The base system uses a query routing procedure to adapt the system to whether or not to go directly to a web search if none of the retrieved documents is relevant to the query. We provide an article-level retrieval mode based on the article database created earlier that can assess whether the system can continue the normal retrieving path. In other words, we first use a retrieval module to retrieve text at the title or summary level, which helps to determine the path to take. The system goes directly to the web search module if we cannot retrieve an article with a similarity score that exceeds a certain threshold. Otherwise, it continues on the traditional path. Then, we retrieve the documents based on the article title we have retrieved before employing reranking to select some of the best candidates to be further forwarded to the prompt. In this step, we set a threshold to remove irrelevant information. The reranker consists of two different paths presenting two ways of calculating the similarity score for a query-document pair; one uses Cohere Multilingual Reranker  6  which outputs a Cohere score; the other works on the named entities and generates a NER score, which is not affected by the OCR errors, thus injecting more stable information into the module. First, we extract the named entities with hmBERT foot_6    [14]  and inject this information into our module. The task involves first creating a string that encompasses the named entities:",
      "section": "Methodology",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Methodology",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "tran2024_104b",
      "id": "tran2024_104b_claim_e80d9200c132",
      "text": "LLaMA3 returning answers in different languages than the input query language creates inconsistent quality in the retrieval augmented generation system.",
      "type": "limitation",
      "paragraph_id": "tran2024_104b_6.2",
      "paragraph": "In Figure  1 , we show an example in which we deliberately damage the query by modifying the name of the exposition from Caros Sandoval to Carlos Sandov. This process is to experiment with whether the system can produce robust results despite OCR errors or user misspellings when querying. It can be seen that, though the Cohere model is obtaining a good answer, its score is relatively low for Figure  1 . However, with the injection of NER into the model, the score is much higher, indicating that more precise information can boost the system's final overall performance. In other cases, Cohere model could still show some robust performances despite the noises we have added and provide high scores without the help of NER information.  The final result (Answer) of LLaMA3 in Figure  2  can be reasonable given the data. However, it can be seen that the answer is returned in different languages, whereas the queries are made only in French. This could create inconsistent quality and should be addressed in the future.",
      "section": "Qualitative Evaluation",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 2,
      "stored_claim_page": 1,
      "stored_claim_section": "Qualitative Evaluation",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "kroll2024_3ce7",
      "id": "kroll2024_3ce7_claim_bfd1622117de",
      "text": "The Comparative Toxigenomics Database knowledge base used for distant labeling is reliable and provides high-quality data for the CDR task.",
      "type": "finding",
      "paragraph_id": "kroll2024_3ce7_10.13",
      "paragraph": "conduct a new search here). The results are shown in Table  7 . For CDR, the different labeling methods seem to perform quite well. The distantly supervised labeling method achieved a similar, but slightly decreased, F1 score compared to the expert labeling. This suggests that the Comparative Toxigenomics Database  [7]  knowledge base used for distant labeling is reliable and provides high-quality data for this task. On ChemProtC, the difference between the labeling methods and expert labeling was more noticeable. The models' performance dropped significantly when trained on noisily generated data, which could also be a cause of our re-grouping of the data. Overall, GPT-4o mode performed the best across most tasks and labeling methods. It consistently outperformed other models with training on BERT models, demonstrating its robustness and effectiveness even when trained on noisily labeled data. This highlights the potential of advanced language models to handle noisy data and achieve high performance without the need for perfect labeling. In brief, LLMs labeled the training data sufficiently well for our purposes and came with acceptable costs in the end.",
      "section": "RQ3: Data Labeling",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 11,
      "stored_claim_page": 1,
      "stored_claim_section": "RQ3: Data Labeling",
      "anchor": "paragraph",
      "audit": {
        "verdict": "partial",
        "title": "A hedge disappeared",
        "explanation": "The paragraph says “This suggests”. The extracted claim states reliability as a fact. The anchor is right, but the claim is stronger than its source.",
        "cue": "This suggests"
      }
    },
    {
      "paper_key": "kroll2024_3ce7",
      "id": "kroll2024_3ce7_claim_82620d172f4a",
      "text": "In digital library implementations, a cheaper model might be favored over a complex model even if the complex model achieves higher accuracy.",
      "type": "finding",
      "paragraph_id": "kroll2024_3ce7_0.2",
      "paragraph": "From a natural language processing perspective, several works exist that propose advanced methods for extracting named entities and their semantic relationships or classifying texts in general; see  [8, 36, 42]  to name just a few. When implementing extraction workflows in a digital library, questions beyond a benchmarkcentric evaluation arise, e.g., about trade-offs between costs and quality. Regarding training and application costs, a cheaper model might be favored over a complex model, achieving higher accuracy. In brief, this work is written from the perspective of a digital library. It differs from existing work in that we 1) compare the trade-off between extraction quality and costs, 2) dive into designing complete end-to-end systems in contrast to benchmark-centric evaluations, and 3) approach how we can generate/retrieve training data.",
      "section": "Introduction",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Introduction",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "kroll2024_3ce7",
      "id": "kroll2024_3ce7_claim_598a7597351d",
      "text": "In distant supervision, if a sentence contains two entities that have a relationship in a knowledge base, the sentence is assumed to express that relationship.",
      "type": "definition",
      "paragraph_id": "kroll2024_3ce7_10.1",
      "paragraph": "Distantly-Supervised Labeling. The related work section describes weak supervision as a possible remedy  [34] . The central idea is that external knowledge is used to label sentences. If some sentence includes two entities and the entities have a relationship within the given knowledge base, then we implicitly assume that the sentence also expresses this relationship. In brief, distant supervision allows fast and large data set generation. That is why we investigate it here. However, it requires external knowledge bases that include the relations someone is interested in, and it might also be limited in precision, as sentences may be labeled in a noisy fashion.",
      "section": "RQ3: Data Labeling",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 11,
      "stored_claim_page": 1,
      "stored_claim_section": "RQ3: Data Labeling",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "kroll2024_3ce7",
      "id": "kroll2024_3ce7_claim_f9b789fb7fce",
      "text": "Named entity recognition typically involves two steps: first identifying entities in text, then disambiguating those text spans to precise identifiers.",
      "type": "definition",
      "paragraph_id": "kroll2024_3ce7_1.0",
      "paragraph": "Named Entity Recognition and Disambiguation. The first step in extracting semantic relationships between named entities is to identify these entities in the text. Usually, recognition tools recognize entities within texts, and subsequent disambiguation tools assign those text spans to precise identifiers to disambiguate them. A comprehensive overview of possible detection methods is given in  [42] . A plethora of different tools exist to identify biomedical entities in texts, e.g., PubTator  [40] , GNormPlus  [41] , GNorm2  [38] , TaggerOne  [27] , and many more. While entity detection is a relevant topic in digital libraries, our work focuses on relation extraction between them and thus assumes that the entities are given.",
      "section": "Related Work",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Related Work",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "yamamoto2025_1935",
      "id": "yamamoto2025_1935_claim_cfbab3e58c02",
      "text": "Semantic relevance between prompts and opinion showed statistically significant differences, with PLAIN showing negative relevance (-0.677) compared to positive relevance in ADJUNCT (0.710) and EXPECTED (0.723).",
      "type": "finding",
      "paragraph_id": "yamamoto2025_1935_21.6",
      "paragraph": "TABLE III: Means and standard deviations of each metric for the online user study. Asterisks (*) indicate metrics for which significant differences across UI conditions were found in the variance analysis. Superscripts P, A, E, and S indicate statistically significant differences in pairwise comparisons with PLAIN, ADJUNCT, EXPECTED, and SCAFFOLDING conditions, respectively. UI Condition Metric PLAIN ADJUNCT EXPECTED SCAFFOLDING Search behavior metrics Task duration (s) 745.8 (476.8) 833.4 (586.1) 869.0 (636.9) 799.0 (480.8) Total SERP duration (s)* 145.4 (120.8) E 264.4 (240.1) 291.3 (256.7) P 285.3 (297.4) Avg. SERP duration per query (s)* 59.3 (95.4) AS 95.3 (66.8) P 96.0 (95.4) 104.8 (86.5) P Number of queries 3.07 (2.42) 3.09 (2.49) 3.81 (2.42) 3.08 (2.43) Number of clickthroughs 6.17 (4.79) 5.94 (3.96) 5.92 (3.33) 6.67 (4.07) Task outcome Semantic relevance between prompts and opinion* -0.677 (0.060) ES 0.710 (0.038) A 0.723 (0.052) A Evaluation of prompts Relevance to theme* -2.59 (2.05) ES 3.04 (2.16) A 3.15 (2.07) A Importance for the theme* -2.46 (2.02) ES 2.96 (2.12) A 3.09 (2.02) A Interestingness* -2.40 (1.99) S 2.88 (2.06) 2.98 (1.95) A Ease of answering -2.59 (2.06) 2.68 (2.03) 2.80 (1.88) Usefulness for deepening understanding* -2.39 (1.96) 2.84 (2.03) 2.94 (1.90) Usefulness for gaining new perspectives* -2.22 (1.85) S 2.65 (1.95) 2.83 (1.90) A Usefulness in organizing information/opinions* -2.31 (1.90) ES 2.81 (2.05) A 2.82 (1.85) A",
      "section": "F. Results",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 9,
      "stored_claim_page": 1,
      "stored_claim_section": "F. Results",
      "anchor": "paragraph",
      "audit": {
        "verdict": "wrong",
        "title": "A dash became a minus sign",
        "explanation": "The audit checked the source table: PLAIN has no prompts, so its cell is a not-applicable dash. The values 0.677, 0.710 and 0.723 belong to ADJUNCT, EXPECTED and SCAFFOLDING. Flattened table text lost the columns. The claimed negative relevance is unsupported.",
        "cue": "-0.677"
      }
    },
    {
      "paper_key": "yamamoto2025_1935",
      "id": "yamamoto2025_1935_claim_041b1c3e9e39",
      "text": "The proposed method uses an LLM to predict opinions that a hypothetical learner might form when reading a web page.",
      "type": "methodology",
      "paragraph_id": "yamamoto2025_1935_27.2",
      "paragraph": "The third limitation involves prompt generation. To address the challenge of not being able to directly observe users' thoughts during a web search, the proposed method employs an LLM to predict opinions that a hypothetical learner might form when reading a page and then generates prompts accordingly. However, because the opinions users form may vary depending on their demographic attributes and prior knowledge, the relevance and effectiveness of the prompts may also differ among users. Thus, the personalization of prompt generation should be considered in future studies.",
      "section": "Limitations and future work",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 3,
      "stored_claim_page": 1,
      "stored_claim_section": "Limitations and future work",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "yamamoto2025_1935",
      "id": "yamamoto2025_1935_claim_06e454365208",
      "text": "In an online study examining inquiry-oriented web search with LLM-based question generation, participants in the SCAFFOLDING condition spent more time per query on the SERP than those in the PLAIN condition.",
      "type": "finding",
      "paragraph_id": "yamamoto2025_1935_25.0",
      "paragraph": "For RQ1, the results of the online study indicate that participants under the SCAFFOLDING condition spent more time per query on the SERP than those under the PLAIN condition. A similar tendency was observed for the ADJUNCT condition. In contrast, although no significant difference was observed in the SERP duration per query under the EXPECTED condition, the total SERP duration was longer than that under the PLAIN condition. These results suggest that under the SCAFFOLDING, EXPECTED, and ADJUNCT conditions, participants may have spent more time on the SERP screen because they reflected on the asked questions or scrutinized the search results prompted by the displayed questions.",
      "section": "VI. DISCUSSION",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 5,
      "stored_claim_page": 1,
      "stored_claim_section": "VI. DISCUSSION",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "bocus2022_ce7f",
      "id": "bocus2022_ce7f_claim_069388bcbc50",
      "text": "The OPERAnet dataset includes data from two rooms, with room '1' being the left room and room '2' being the right room as shown in Figure 1.",
      "type": "finding",
      "paragraph_id": "bocus2022_ce7f_10.23",
      "paragraph": "• room_no: room ID specified as \"1\" (left room in Fig.  1 ) or \"2\" (right room in Fig.  1 ).",
      "section": "Data Records",
      "page": 8,
      "bbox_source": "docling",
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Data Records",
      "anchor": "paragraph",
      "audit": {
        "verdict": "correct",
        "title": "The source supports the statement",
        "explanation": "The claim preserves the stored paragraph’s room identifiers and their left/right positions in Figure 1. This pair was judged correct in the historical audit.",
        "cue": "left room"
      }
    },
    {
      "paper_key": "bocus2022_ce7f",
      "id": "bocus2022_ce7f_claim_04944cb19a12",
      "text": "In the OPERAnet dataset, device-free dynamic localization experiments use a CSI transmitter labeled NUC3 and a CSI receiver labeled NUC2.",
      "type": "methodology",
      "paragraph_id": "bocus2022_ce7f_6.0",
      "paragraph": "Device-free dynamic localization. CSI transmitter (NUC3) and CSI receiver (NUC2) are placed side by side and the target moves along a short straight path for each experiment number.",
      "section": "exp044-exp048",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "exp044-exp048",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "bocus2022_ce7f",
      "id": "bocus2022_ce7f_claim_4ea99acc4119",
      "text": "Artificial intelligence algorithms can be used to infer the number of people in an environment using UWB and WiFi sensor parameters.",
      "type": "methodology",
      "paragraph_id": "bocus2022_ce7f_10.33",
      "paragraph": "Considering the crowd counting experiment, Fig.  6  shows the first path power level (in dBm) for the two UWB systems between a given pair of nodes in each case. The first path power level (fp_pow_dbm) has been computed using the formula given in the DW1000 manual  44  . As can be observed, the first path power level increases gradually as each person was moving out of the monitoring area. This is an expected behaviour since the LoS signal becomes less and less obstructed. By using the fp_pow_dbm parameter together with other parameters such as overall received UWB signal power level (rx_pow_dbm), UWB CIR data and WiFi CSI data, the number of people in a given environment can be inferred through the use of artificial intelligence algorithms.",
      "section": "Data Records",
      "page": 2,
      "bbox_source": "docling",
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Data Records",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "janssens2024_2f53",
      "id": "janssens2024_2f53_claim_054a1907ff89",
      "text": "Two separately trained polynomial regression models are used: one for when a rail vehicle is present and one for when it is absent.",
      "type": "methodology",
      "paragraph_id": "janssens2024_2f53_1.0",
      "paragraph": "We demonstrate the use of two separately trained polynomial regression models, when a rail vehicle is present and absent, in order to perform crowd size estimation using the change in the RSSI between wireless sensor nodes in the different environment states.",
      "section": "•",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "•",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "janssens2024_2f53",
      "id": "janssens2024_2f53_claim_07adaab5862f",
      "text": "In device-free crowd size estimation on subway platforms when a vehicle is present, the mean absolute error (MAE) is 4.478 people.",
      "type": "finding",
      "paragraph_id": "janssens2024_2f53_9.7",
      "paragraph": "When evaluating the crowd size estimation with no vehicle present, a median error of 2.769 people is observed, along with an MAE of 3.342 people and an RMSE of 4.211. The CDF plot is shown in Figure  8  as a green dash-dotted line.   When evaluating the crowd size estimation when a vehicle is present, a median error of 3.304 people is observed, along with an MAE of 4.478 people and an RMSE of 6.326. The CDF plot is shown in Figure  8  as an orange dashed line.",
      "section": "Crowd Size Estimation",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 15,
      "stored_claim_page": 1,
      "stored_claim_section": "Crowd Size Estimation",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "janssens2024_2f53",
      "id": "janssens2024_2f53_claim_09a6e0345c6f",
      "text": "The combined model for device-free crowd size estimation on subway platforms achieves a root mean square error (RMSE) of 4.706.",
      "type": "finding",
      "paragraph_id": "janssens2024_2f53_10.1",
      "paragraph": "When we look at more statistical metrics, we obtain a median error of 2.856 people, an MAE of 3.567 people, and an RMSE of 4.706. These numbers are a clear improvement over the use of a single polynomial regression model trained on both states, which results in a median error of 4.617 people, an MAE of 6.192 people, and an RMSE of 8.250.",
      "section": "Combined Model",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 6,
      "stored_claim_page": 1,
      "stored_claim_section": "Combined Model",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "tai2024_5a71",
      "id": "tai2024_5a71_claim_151f5dfa0b31",
      "text": "BERT models fine-tuned with OCR datasets show greater resilience to OCR noise in classification tasks than models pretrained on born-digital texts.",
      "type": "finding",
      "paragraph_id": "tai2024_5a71_8.8",
      "paragraph": "Jiang et al.  [19]  created large-scale parallel datasets of OCR'd text and human proofread counterparts sourced from Project Gutenberg and Hathitrust Digital Library, containing over 19,000 works in six domains: fiction, social science, agriculture, world war history, medicine, and business. With this benchmark dataset, Jiang  [18]  evaluated domain classification of text with OCR errors. The BERT models fine-tuned with their dataset show better encoding stability and resilience to OCR noise in classification tasks than models pretrained on born-digital texts. To evaluate the impact of OCR noise, Jiang et al.  [20]  encoded OCR'd and human-corrected versions of book chapters with pre-trained and fine-tuned BERT models and then compared them for similarities and quality of encodings. The authors' evaluation showed that BERT embeddings can be resilient to OCR errors when encoding chapter-level content with high NDCG scores. When encoding word and sentence level content, OCR errors can introduce erroneous tokens and disrupt the coherence of sentences.",
      "section": "Optical Character Recognition",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 13,
      "stored_claim_page": 1,
      "stored_claim_section": "Optical Character Recognition",
      "anchor": "paragraph",
      "audit": {
        "verdict": "partial",
        "title": "One dataset became a general rule",
        "explanation": "The paragraph attributes the result to Jiang and a particular dataset. The extraction broadens this to OCR datasets in general and loses the attribution.",
        "cue": "with their dataset"
      }
    },
    {
      "paper_key": "tai2024_5a71",
      "id": "tai2024_5a71_claim_0906493ec965",
      "text": "Three major areas of interest were identified in AI applications for libraries: recommendation systems, information and resource retrieval, and optical character recognition.",
      "type": "finding",
      "paragraph_id": "tai2024_5a71_3.0",
      "paragraph": "In this section, we explore RQ2 (\"How is AI currently being researched and applied in libraries?\") and RQ3 (\"What are the limitations and future research directions of different AI technologies as they relate to libraries?\"). This section consists of a detailed investigation of the latest research on practical applications of artificial intelligence in libraries. Through our review, we identified three major areas of interest: recommendation systems, information and resource retrieval, and optical character recognition. In each subsection, we outline different papers, emphasizing the technologies utilized, the limitations of the research, and directions for future studies. Table  1  shows the papers we reviewed for this section, categorized by the three major areas of interest and sub-areas of research.",
      "section": "Applications and Research Directions",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "Applications and Research Directions",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "tai2024_5a71",
      "id": "tai2024_5a71_claim_08b63849aa90",
      "text": "Hall and McKee identified extensive opportunities for libraries to use prompt engineering with ChatGPT in tasks like summarizing content and developing curricula and rubrics.",
      "type": "finding",
      "paragraph_id": "tai2024_5a71_6.1",
      "paragraph": "Traditionally, reference services are provided through face-toface conversations. However, during times of congestion, nonoperational hours, and for users who want to access resources remotely  [2] , AI-operated reference services could be particularly useful. Chatbots offer readily accessible information and personalized user assistance, ideally suited for library reference services. Many researchers have found ChatGPT particularly useful to help patrons navigate through the library's reserve of information and resources. Hall and McKee  [13]  and Lo  [25]  also observed extensive opportunities for libraries to use prompt engineering with Chat-GPT in tasks like summarizing content and developing curricula and rubrics. With effective prompting of novel language models, librarians can offer personalized assistance to patrons, helping with their research efforts and improving users' information literacy skills. Adetayo  [2]  argued for the integration of BingChat into library websites and catalogs to reshape existing digital reference services. Although BingChat and other industry chatbots are extremely resourceful, they are not trained specifically for libraries, so they may not be able to answer questions pertinent to specific library information and their unique resources.",
      "section": "Chatbots.",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 2,
      "stored_claim_page": 1,
      "stored_claim_section": "Chatbots.",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "ingram2025_642d",
      "id": "ingram2025_642d_claim_00d7b853d1ec",
      "text": "Two locally hosted LLMs were used to assign binary relevance labels (Relevant or Non-Relevant) to abstracts in the LLM filtering process.",
      "type": "methodology",
      "paragraph_id": "ingram2025_642d_9.0",
      "paragraph": "We use two locally hosted LLMs to assign a binary relevance label (Relevant or Non-Relevant) to each abstract based on whether or not it describes a meaningful contribution to the given SDG targets. The LLMs were given identical prompts that include instructions to return a binary label along with a brief justification. The process is described in Section IV-A.",
      "section": "D. LLM Filtering",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 1,
      "stored_claim_page": 1,
      "stored_claim_section": "D. LLM Filtering",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "ingram2025_642d",
      "id": "ingram2025_642d_claim_0e0a9a4cd3bf",
      "text": "Feature inspection or semantic embedding comparisons would be required to characterize the specific lexical or conceptual criteria each model implicitly applies.",
      "type": "methodology",
      "paragraph_id": "ingram2025_642d_19.2",
      "paragraph": "Fig.  5  confirms that all three classifiers outperform chance. The above-baseline performance indicates that disagreement is non-random and tied to consistent lexical differences, though it does not imply that either model applies a single coherent or interpretable criterion. Further analysis, such as feature inspection or semantic embedding comparisons, would be required to characterize the specific lexical or conceptual criteria each model implicitly applies.",
      "section": "D. Learnability of Filtering Behavior",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 2,
      "stored_claim_page": 1,
      "stored_claim_section": "D. Learnability of Filtering Behavior",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "ingram2025_642d",
      "id": "ingram2025_642d_claim_0631056d69b7",
      "text": "TF-IDF vectors are used as input features for training a logistic regression classifier to predict document relevance labels assigned by different models.",
      "type": "methodology",
      "paragraph_id": "ingram2025_642d_14.1",
      "paragraph": "Using TF-IDF vectors as input features, we train a logistic regression classifier to predict which model labeled each document as relevant. We evaluate performance using five-fold cross-validation and report the area under the ROC curve (AUC) as the evaluation metric. All models are trained separately for each SDG to isolate domain-specific patterns and prevent topic leakage across goals.",
      "section": "D. Learnability of Filtering Differences",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 2,
      "stored_claim_page": 1,
      "stored_claim_section": "D. Learnability of Filtering Differences",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "zhang2024_aea3",
      "id": "zhang2024_aea3_claim_10eb06633adf",
      "text": "Converting a single LLM call that sequentially generates introductions for N researchers into N parallel calls, each generating introduction for one researcher, eliminates the 'confuse context' hallucination.",
      "type": "finding",
      "paragraph_id": "zhang2024_aea3_6.0",
      "paragraph": "After filtering, as all of the remaining documents are highly related to the query, the answer generation task is simplified to summarizing all given documents, which in our case should produce short introduction for each researcher. In this stage, we observed that LLMs are prone to confuse information among different documents, e.g. attributing researcher A's outcome to researcher B in the answer, which is totally unacceptable. This is also a typical kind of hallucination observed in previous work  [12] . Since all documents have the corresponding researcher name, we first tried to add a preprocessing that sorts and integrates documents according to researcher identity before generation. The hallucination reduces but still exists. Considering that each integrated context now contains full information for one researcher, similar to filtering, we converted one LLM call that sequentially generates introduction for all 𝑁 researchers to 𝑁 parallel calls that each generates introduction for one researcher and merged the outputs at the end. This not only makes each task easier and eliminates the \"confuse context\" hallucination since they are isolated, but also accelerates the generation as explained before.",
      "section": "Generation",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 6,
      "stored_claim_page": 1,
      "stored_claim_section": "Generation",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "zhang2024_aea3",
      "id": "zhang2024_aea3_claim_23d4516e921c",
      "text": "For knowledge distillation to work effectively, a LLM must be able to stably generate outputs of expected quality to serve as a good teacher model.",
      "type": "hypothesis",
      "paragraph_id": "zhang2024_aea3_5.7",
      "paragraph": "The advantage is that it requires no or seldom handcrafted supervised data, which saves a lot of human labor. The premise is that there must exist a LLM that can stably generate outputs of expected quality, otherwise the \"student\" won't have a good \"textbook\" to learn from. That's why GPT-4, one of the best-performing LLM, is widely used for this task. In our case, we experimented with the largest LLM among open-source families to choose the best teacher model, and also tried prompt engineering techniques including few-shot  [3]  and Chain-of-Thought (CoT)  [24]  prompting to further improve data quality. Few-shot prompting, also called in-context learning, provides several input and output examples of the task before asking the model to solve a new problem. This gives the model better understanding of a specific task. CoT prompts the LLM to break down question and output intermediate thinking step, which can make the final answer more accurate. In the case of relevance judgement, we have found that instructing LLM to additionally explain the reason instead of just giving the conclusion leads to better accuracy.",
      "section": "Retrieval and Filter",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 12,
      "stored_claim_page": 1,
      "stored_claim_section": "Retrieval and Filter",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "zhang2024_aea3",
      "id": "zhang2024_aea3_claim_03030103ce77",
      "text": "In the indexing process for retrieval-augmented generation systems, keywords are used to represent the academic information from proposals or abstracts.",
      "type": "methodology",
      "paragraph_id": "zhang2024_aea3_4.2",
      "paragraph": "These keywords covered most of the academic information from the proposal or abstract.",
      "section": "Indexing",
      "page": 1,
      "bbox_source": null,
      "coordinate_candidates": 2,
      "stored_claim_page": 1,
      "stored_claim_section": "Indexing",
      "anchor": "paragraph",
      "audit": null
    },
    {
      "paper_key": "bocus2022_ce7f",
      "id": "claim-0a293d9d",
      "text": "First multimodal dataset combining RF (WiFi CSI, Passive WiFi Radar, UWB) and vision (Kinect) modalities, time-synchronized, intended jointly for HAR and passive (non-cooperative) indoor localization.",
      "type": "reading result",
      "section": "Background & Summary / Contributions",
      "page": 2,
      "anchor": "paper",
      "paragraph": null,
      "paragraph_id": null,
      "bbox_source": null,
      "coordinate_candidates": null,
      "stored_claim_page": 2,
      "stored_claim_section": "Background & Summary / Contributions",
      "audit": null
    }
  ],
  "terms": [
    {
      "id": "Method:5-fold-cross-validation",
      "slug": "5-fold-cross-validation",
      "name": "5-fold cross-validation",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:activity-recognition",
      "slug": "activity-recognition",
      "name": "Activity Recognition",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateMethod:automatic-device-mapping",
      "slug": "automatic-device-mapping",
      "name": "Automatic device mapping",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:bert-score",
      "slug": "bert-score",
      "name": "BERT score",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:bm25",
      "slug": "bm25",
      "name": "BM25",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:csi-extraction",
      "slug": "csi-extraction",
      "name": "CSI Extraction",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Hardware:camera-sensor",
      "slug": "camera-sensor",
      "name": "Camera",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Method:cnn",
      "slug": "cnn",
      "name": "Convolutional Neural Network",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        },
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:crowd-counting",
      "slug": "crowd-counting",
      "name": "Crowd Counting",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        },
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Problem:density-estimation",
      "slug": "density-estimation",
      "name": "Crowd Density Estimation",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Problem:crowd-monitoring",
      "slug": "crowd-monitoring",
      "name": "Crowd Monitoring",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Method:deep-learning",
      "slug": "deep-learning",
      "name": "Deep Learning",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:device-free-localization",
      "slug": "device-free-localization",
      "name": "Device-Free Localization",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        },
        {
          "key": "janssens2024_2f53",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateProblem:document-classification",
      "slug": "document-classification",
      "name": "Document Classification",
      "type": "Problem",
      "status": "candidate",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateMethod:document-classification",
      "slug": "document-classification",
      "name": "Document classification",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:fall-detection",
      "slug": "fall-detection",
      "name": "Fall Detection",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Method:few-shot-learning",
      "slug": "few-shot-learning",
      "name": "Few-Shot Learning",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:few-shot-prompting",
      "slug": "few-shot-prompting",
      "name": "Few-shot prompting",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_METHOD"
        },
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        },
        {
          "key": "yamamoto2025_1935",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:fine-tuning",
      "slug": "fine-tuning",
      "name": "Fine-Tuning",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateHardware:gpu",
      "slug": "gpu",
      "name": "GPU",
      "type": "Hardware",
      "status": "candidate",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_HARDWARE"
        },
        {
          "key": "zhang2024_aea3",
          "relation": "USES_HARDWARE"
        },
        {
          "key": "ingram2025_642d",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Problem:gesture-recognition",
      "slug": "gesture-recognition",
      "name": "Gesture Recognition",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Method:human-activity-recognition",
      "slug": "human-activity-recognition",
      "name": "Human Activity Recognition",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:indoor-localization",
      "slug": "indoor-localization",
      "name": "Indoor Localization",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        },
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Hardware:intel-5300",
      "slug": "intel-5300",
      "name": "Intel 5300",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Hardware:laptop",
      "slug": "laptop",
      "name": "Laptop",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "CandidateMethod:large-language-models-llms",
      "slug": "large-language-models-llms",
      "name": "Large Language Models (LLMs)",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        },
        {
          "key": "yamamoto2025_1935",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Hardware:lora",
      "slug": "lora",
      "name": "LoRa",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Method:logistic-regression",
      "slug": "logistic-regression",
      "name": "Logistic Regression",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        },
        {
          "key": "janssens2024_2f53",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:motion-detection",
      "slug": "motion-detection",
      "name": "Motion Detection",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Problem:multipath-fading",
      "slug": "multipath-fading",
      "name": "Multipath Fading",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateMethod:named-entity-recognition",
      "slug": "named-entity-recognition",
      "name": "Named Entity Recognition",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:natural-language-processing",
      "slug": "natural-language-processing",
      "name": "Natural Language Processing",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_METHOD"
        },
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Dataset:operanet",
      "slug": "operanet",
      "name": "OPERAnet",
      "type": "Dataset",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "EVALUATES_ON"
        }
      ]
    },
    {
      "id": "CandidateMethod:optical-character-recognition",
      "slug": "optical-character-recognition",
      "name": "Optical Character Recognition",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:passive-wifi-radar",
      "slug": "passive-wifi-radar",
      "name": "Passive WiFi Radar",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Hardware:passive-wifi-radar",
      "slug": "passive-wifi-radar",
      "name": "Passive WiFi Radar",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Problem:people-counting",
      "slug": "people-counting",
      "name": "People Counting",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateMethod:prompt-engineering",
      "slug": "prompt-engineering",
      "name": "Prompt engineering",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        },
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:rssi-fingerprinting",
      "slug": "rssi-fingerprinting",
      "name": "RSSI Fingerprinting",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:random-forest",
      "slug": "random-forest",
      "name": "Random Forest",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:received-signal-strength",
      "slug": "received-signal-strength",
      "name": "Received Signal Strength (RSS)",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:reranking",
      "slug": "reranking",
      "name": "Reranking",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:retrieval-augmented-generation",
      "slug": "retrieval-augmented-generation",
      "name": "Retrieval-Augmented Generation",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "zhang2024_aea3",
          "relation": "USES_METHOD"
        },
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:sensor-fusion",
      "slug": "sensor-fusion",
      "name": "Sensor Fusion",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Problem:sign-language-recognition",
      "slug": "sign-language-recognition",
      "name": "Sign Language Recognition",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Method:spectrogram",
      "slug": "spectrogram",
      "name": "Spectrogram",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:supervised-learning",
      "slug": "supervised-learning",
      "name": "Supervised Learning",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:tf-idf",
      "slug": "tf-idf",
      "name": "TF-IDF",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Hardware:uwb",
      "slug": "uwb",
      "name": "Ultra-Wideband",
      "type": "Hardware",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_HARDWARE"
        }
      ]
    },
    {
      "id": "Method:wifi-csi-sensing",
      "slug": "wifi-csi-sensing",
      "name": "WiFi CSI Sensing",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "Method:xgboost",
      "slug": "xgboost",
      "name": "XGBoost",
      "type": "Method",
      "status": "curated",
      "papers": [
        {
          "key": "kroll2024_3ce7",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateMethod:binary-classification",
      "slug": "binary-classification",
      "name": "binary classification",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateProblem:binary-classification",
      "slug": "binary-classification",
      "name": "binary classification",
      "type": "Problem",
      "status": "candidate",
      "papers": [
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateMethod:cosine-similarity",
      "slug": "cosine-similarity",
      "name": "cosine similarity",
      "type": "Method",
      "status": "candidate",
      "papers": [
        {
          "key": "tai2024_5a71",
          "relation": "USES_METHOD"
        },
        {
          "key": "ingram2025_642d",
          "relation": "USES_METHOD"
        },
        {
          "key": "tran2024_104b",
          "relation": "USES_METHOD"
        },
        {
          "key": "yamamoto2025_1935",
          "relation": "USES_METHOD"
        }
      ]
    },
    {
      "id": "CandidateProblem:information-retrieval",
      "slug": "information-retrieval",
      "name": "information retrieval",
      "type": "Problem",
      "status": "candidate",
      "papers": [
        {
          "key": "tai2024_5a71",
          "relation": "ADDRESSES"
        },
        {
          "key": "ingram2025_642d",
          "relation": "ADDRESSES"
        },
        {
          "key": "tran2024_104b",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "Problem:multimodal-sensor-fusion",
      "slug": "multimodal-sensor-fusion",
      "name": "multimodal sensor fusion",
      "type": "Problem",
      "status": "curated",
      "papers": [
        {
          "key": "bocus2022_ce7f",
          "relation": "ADDRESSES"
        }
      ]
    },
    {
      "id": "CandidateProblem:overfitting",
      "slug": "overfitting",
      "name": "overfitting",
      "type": "Problem",
      "status": "candidate",
      "papers": [
        {
          "key": "tai2024_5a71",
          "relation": "ADDRESSES"
        },
        {
          "key": "janssens2024_2f53",
          "relation": "ADDRESSES"
        }
      ]
    }
  ],
  "curation_terms": [
    "CandidateMethod:bm25",
    "CandidateMethod:bert-score",
    "CandidateMethod:automatic-device-mapping"
  ],
  "start_claim": "tran2024_104b_claim_5011d54ce58b"
}
