<?xml version="1.0" encoding="utf-8"?>
<?xml-stylesheet type="text/xsl" href="xsl/oai2.xslt"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-09-21T19:34:19Z</responseDate>
  <request verb="GetRecord" metadataPrefix="xMetaDissPlus" identifier="oai:kobv.de-opus4-uni-passau:293">https://opus4.kobv.de/opus4-uni-passau/oai</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:kobv.de-opus4-uni-passau:293</identifier>
        <datestamp>2025-08-13</datestamp>
        <setSpec>bibliography:false</setSpec>
        <setSpec>doc-type:PhDThesis</setSpec>
        <setSpec>open_access</setSpec>
        <setSpec>ddc</setSpec>
        <setSpec>ddc:004</setSpec>
      </header>
      <metadata>
        <xMetaDiss:xMetaDiss xmlns:xMetaDiss="http://www.d-nb.de/standards/xmetadissplus/" xmlns:cc="http://www.d-nb.de/standards/cc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:dcmitype="http://purl.org/dc/dcmitype/" xmlns:dcterms="http://purl.org/dc/terms/" xmlns:pc="http://www.d-nb.de/standards/pc/" xmlns:urn="http://www.d-nb.de/standards/urn/" xmlns:hdl="http://www.d-nb.de/standards/hdl/" xmlns:doi="http://www.d-nb.de/standards/doi/" xmlns:thesis="http://www.ndltd.org/standards/metadata/etdms/1.0/" xmlns:ddb="http://www.d-nb.de/standards/ddb/" xmlns:dini="http://www.d-nb.de/standards/xmetadissplus/type/" xmlns="http://www.d-nb.de/standards/subject/" xsi:schemaLocation="http://www.d-nb.de/standards/xmetadissplus/ https://d-nb.info/standards/schema/xmetadissplus.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
          <dc:title xsi:type="ddb:titleISO639-2" lang="eng">Usage-Driven Unified Model for User Profile and Data Source Profile Extraction</dc:title>
          <dc:creator xsi:type="pc:MetaPers">
            <pc:person>
              <pc:name type="nameUsedByThePerson">
                <pc:foreName>Lyes</pc:foreName>
                <pc:surName>Limam</pc:surName>
              </pc:name>
              <pc:academicTitle>Dr.</pc:academicTitle>
            </pc:person>
          </dc:creator>
          <dc:subject xsi:type="xMetaDiss:DDC-SG">004</dc:subject>
          <dc:subject xsi:type="xMetaDiss:SWD">Information Retrieval</dc:subject>
          <dc:subject xsi:type="xMetaDiss:SWD">Suchmaschine</dc:subject>
          <dc:subject xsi:type="xMetaDiss:SWD">Suchverfahren</dc:subject>
          <dc:subject xsi:type="xMetaDiss:SWD">Abfrageverarbeitung</dc:subject>
          <dcterms:abstract xsi:type="ddb:contentISO639-2" ddb:type="noScheme" lang="eng">This thesis addresses a problem related to usage analysis in information retrieval&#13;
systems. Indeed, we exploit the history of search queries as support of analysis to&#13;
extract a profile model. The objective is to characterize the user and the data source&#13;
that interact in a system to allow different types of comparison (user-to-user, sourceto-&#13;
source, user-to-source). According to the study we conducted on the work done on&#13;
profile model, we concluded that the large majority of the contributions are strongly&#13;
related to the applications within they are proposed. As a result, the proposed&#13;
profile models are not reusable and suffer from several weaknesses. For instance,&#13;
these models do not consider the data source, they lack of semantic mechanisms and&#13;
they do not deal with scalability (in terms of complexity). Therefore, we propose&#13;
a generic model of user and data source profiles. The characteristics of this model&#13;
are the following. First, it is generic, being able to represent both the user and the&#13;
data source. Second, it enables to construct the profiles in an implicit way based on histories of search queries. Third, it defines the profile as a set of topics of interest,&#13;
each topic corresponding to a semantic cluster of keywords extracted by a specific&#13;
clustering algorithm. Finally, the profile is represented according to the vector space&#13;
model. The model is composed of several components organized in the form of a&#13;
framework, in which we assessed the complexity of each component.&#13;
The main components of the framework are:&#13;
• a method for keyword queries disambiguation&#13;
• a method for semantically representing search query logs in the form of a&#13;
taxonomy;&#13;
• a clustering algorithm that allows fast and efficient identification of topics of&#13;
interest as semantic clusters of keywords;&#13;
• a method to identify user and data source profiles according to the generic&#13;
model.&#13;
&#13;
This framework enables in particular to perform various tasks related to usage-based&#13;
structuration of a distributed environment. As an example of application, the framework&#13;
is used to the discovery of user communities, and the categorization of data&#13;
sources. To validate the proposed framework, we conduct a series of experiments&#13;
on real logs from the search engine AOL search, which demonstrate the efficiency&#13;
of the disambiguation method in short queries, and show the relation between the&#13;
quality based clustering and the structure based clustering.</dcterms:abstract>
          <dcterms:abstract xsi:type="ddb:contentISO639-2" ddb:type="noScheme" lang="ger">Die Arbeit befasst sich mit der Nutzungsanalyse von Informationssuchsystemen.&#13;
Auf Basis vergangener Anfragen sollen Nutzungsprofile ermittelt werden. Diese Profile&#13;
charakterisieren die im Netz interagierenden Anwender und Datenquellen und&#13;
ermöglichen somit Vergleiche von Anwendern, Anwendern und Datenquellen wie&#13;
auch Vergleiche von Datenquellen. Die Arbeit am Profil-Modell und die damit verbundenen&#13;
Studien zeigten, dass praktisch alle Beiträge stark auf die entsprechende&#13;
Anwendung angepasst sind. Als Ergebnis sind die vorgeschlagenen Profil-Modelle&#13;
nicht wiederverwendbar; darüber hinaus weisen sie mehrere Schwächen auf. Die&#13;
Modelle sind zum Beispiel nicht für Datenquellen einsetzbar, Mechanismen für semantische&#13;
Analysen sind nicht vorhanden oder sie verfügen übe keine adequate&#13;
Skalierbarkeit (Komplexität). Um das Ziel von Nutzerprofilen zu erreichen wurde&#13;
ein einheitliches Modell entwickelt. Dies ermöglicht die Modellierung von beiden Elementen:&#13;
Nutzerprofilen und Datenquellen. Ein solches Nutzerprofil wird als Menge&#13;
von Themenbereichen definiert, welche das Verhalten des Anwenders (Suchanfragen)&#13;
beziehungsweise die Inhalte der Datenquelle charakterisieren. Das Modell ermöglicht&#13;
die automatische Profilerstellung auf Basis der vergangenen Suchanfragen, welches&#13;
unmittelbar zur Verfügung steht. Jeder Themenbereich korrespondiert einem Cluster&#13;
von Schlüsselwörtern, die durch einen semantischen Clustering-Algorithmus extrahiert&#13;
werden. Das Modell umfasst mehrere Komponenten, welche als Framework&#13;
strukturiert sind. Die Komplexität jeder einzelner Komponente ist dabei festgehalten&#13;
worden. Die wichtigsten Komponenten sind die Folgenden:&#13;
• eine Methode zur Anfragen Begriffsklärung&#13;
• eine Methode zur semantischen Darstellung der Logs als Taxonomie&#13;
• einen Cluster-Algorithmus, der Themenbereiche (Anwender-Interessen,&#13;
Datenquellen-Inhalte) über semantische Cluster der Schlüsselbegriffe identifiziert&#13;
• eine Methode zur Berechnung des Nutzerprofils und des Profils der Datenquellen&#13;
ausgehend von einem einheitlichen Modell&#13;
Als Beispiel der vielfältigen Einsatzmöglichkeiten hinsichtlich Nutzerprofilen wurde&#13;
das Framework abschließend auf zwei Beispiel-Szenarien angewendet: die Ermittlung&#13;
von Anwender-Communities und die Kategorisierung von Datenquellen. Das&#13;
Framework wurde durch Experimente validiert, welche auf Suchanfrage-Logs von&#13;
AOL Search basieren. Die Effizienz der Verfahren wurde für kleine Anfragen demonstriert&#13;
und zeigt die Beziehung zwischen dem Qualität-basiertem Clustering und dem&#13;
Struktur-basiertem Clustering.</dcterms:abstract>
          <dcterms:abstract xsi:type="ddb:contentISO639-2" ddb:type="noScheme" lang="fre">La problématique traitée dans la thèse s’inscrit dans le cadre de l’analyse d’usage&#13;
dans les systèmes de recherche d’information. En effet, nous nous intéressons à&#13;
l’utilisateur à travers l’historique de ses requêtes, utilisées comme support d’analyse&#13;
pour l’extraction d’un profil d’usage. L’objectif est de caractériser l’utilisateur et les&#13;
sources de données qui interagissent dans un réseau afin de permettre des comparaisons&#13;
utilisateur-utilisateur, source-source et source-utilisateur. Selon une étude que&#13;
nous avons menée sur les travaux existants sur les modèles de profilage, nous avons&#13;
conclu que la grande majorité des contributions sont fortement liés aux applications&#13;
dans lesquelles ils étaient proposés. En conséquence, les modèles de profils proposés&#13;
ne sont pas réutilisables et présentent plusieurs faiblesses. Par exemple, ces modèles&#13;
ne tiennent pas compte de la source de données, ils ne sont pas dotés de mécanismes&#13;
de traitement sémantique et ils ne tiennent pas compte du passage à l’échelle (en&#13;
termes de complexité). C’est pourquoi, nous proposons dans cette thèse un modèle&#13;
d’utilisateur et de source de données basé sur l’analyse d’usage. Les caractéristiques&#13;
de ce modèle sont les suivantes. Premièrement, il est générique, permettant&#13;
de représenter à la fois un utilisateur et une source de données. Deuxièmement,&#13;
il permet de construire le profil de manière implicite à partir de l’historique de requêtes&#13;
de recherche. Troisièmement, il définit le profil comme un ensemble de centres&#13;
d’intérêts, chaque intérêt correspondant à un cluster sémantique de mots-clés déterminé&#13;
par un algorithme de clustering spécifique. Et enfin, dans ce modèle le profil&#13;
est représenté dans un espace vectoriel. Les différents composants du modèle sont&#13;
organisés sous la forme d’un framework, la complexité de chaque composant y est&#13;
evaluée. Le framework propose :&#13;
• une methode pour la désambiguisation de requêtes ;&#13;
• une méthode pour la représentation sémantique des logs sous la forme d’une&#13;
taxonomie ;&#13;
• un algorithme de clustering qui permet l’identification rapide et efficace des centres d’intérêt représentés par des clusters sémantiques de mots clés ;&#13;
• une méthode pour le calcul du profil de l’utilisateur et du profil de la source&#13;
de données à partir du modèle générique.&#13;
Le framework proposé permet d’effectuer différentes tâches liées à la structuration&#13;
d’un environnement distribué d’un point de vue usage. Comme exemples&#13;
d’application, le framework est utilisé pour la découverte de communautés d’utilisateurs&#13;
et la catégorisation de sources de données. Pour la validation du framework,&#13;
une série d’expérimentations est menée en utilisant des logs du moteur de recherche&#13;
AOL-search, qui ont démontrées l’efficacité de la désambiguisation sur des requêtes&#13;
courtes, et qui ont permis d’identification de la relation entre le clustering basé sur&#13;
une fonction de qualité et le clustering basé sur la structure.</dcterms:abstract>
          <dc:publisher xsi:type="cc:Publisher" type="dcterms:ISO3166">
            <cc:universityOrInstitution>
              <cc:name>Universität Passau</cc:name>
              <cc:place>Passau</cc:place>
            </cc:universityOrInstitution>
            <cc:address cc:Scheme="DIN5008">Innstrasse 29, 94032 Passau</cc:address>
          </dc:publisher>
          <dc:contributor xsi:type="pc:Contributor" type="dcterms:ISO3166" thesis:role="advisor">
            <pc:person>
              <pc:name type="nameUsedByThePerson">
                <pc:foreName>Harald</pc:foreName>
                <pc:surName>Kosch</pc:surName>
              </pc:name>
            </pc:person>
          </dc:contributor>
          <dc:contributor xsi:type="pc:Contributor" type="dcterms:ISO3166" thesis:role="advisor">
            <pc:person>
              <pc:name type="nameUsedByThePerson">
                <pc:foreName>Lionel</pc:foreName>
                <pc:surName>Brunie</pc:surName>
              </pc:name>
            </pc:person>
          </dc:contributor>
          <dcterms:dateAccepted xsi:type="dcterms:W3CDTF">2014-06-24</dcterms:dateAccepted>
          <dcterms:issued xsi:type="dcterms:W3CDTF">2015-07-09</dcterms:issued>
          <dc:type xsi:type="dini:PublType">PhDThesis</dc:type>
          <dc:type xsi:type="dcterms:DCMIType">Text</dc:type>
          <dc:identifier xsi:type="urn:nbn">urn:nbn:de:bvb:739-opus4-2936</dc:identifier>
          <dcterms:medium xsi:type="dcterms:IMT">application/pdf</dcterms:medium>
          <dc:language xsi:type="dcterms:ISO639-2">eng</dc:language>
          <dc:rights>Standardbedingung laut Einverständniserklärung</dc:rights>
          <thesis:degree>
            <thesis:level>thesis.doctoral</thesis:level>
            <thesis:grantor xsi:type="cc:Corporate">
              <cc:universityOrInstitution>
                <cc:name>Universität Passau</cc:name>
                <cc:place>Passau</cc:place>
                <cc:department>
                  <cc:name>Fakultät für Informatik und Mathematik</cc:name>
                </cc:department>
              </cc:universityOrInstitution>
            </thesis:grantor>
          </thesis:degree>
          <ddb:contact ddb:contactID="F6000-0384"/>
          <ddb:fileNumber>1</ddb:fileNumber>
          <ddb:fileProperties ddb:fileName="Doktorarbeit_Lyes_Limam.pdf" ddb:fileSize="3946622" ddb:fileID="file293-0"/>
          <ddb:transfer ddb:type="dcterms:URI">https://opus4.kobv.de/opus4-uni-passau/oai/container/index/docId/293</ddb:transfer>
          <ddb:identifier ddb:type="URL">https://opus4.kobv.de/opus4-uni-passau/frontdoor/index/index/docId/293</ddb:identifier>
          <ddb:rights ddb:kind="free"/>
        </xMetaDiss:xMetaDiss>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
