<?xml version="1.0" encoding="UTF-8"?>
<CourseUnit xmlns="http://www.manchester.ac.uk/CUICourseUnitDetails" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.manchester.ac.uk/CUICourseUnitDetails.xsd">
  <UnitCode Applicant="Y" Label="Unit code" Student="Y">
    <Code>SOST30071</Code>
  </UnitCode>
  <UnitTitle Applicant="Y" Label="Unit title" Student="Y">
    <Title>Quantitative Text Analysis in the Social Sciences</Title>
  </UnitTitle>
  <MaxUnits Applicant="Y" Label="Credit rating" Student="Y">
    <Units>20</Units>
  </MaxUnits>
  <TeachingPeriods Applicant="Y" Label="Teaching period(s)" Student="Y">
    <Period>Semester 1</Period>
  </TeachingPeriods>
  <AcademicCareer Applicant="Y" Label="Academic career" Student="Y">
    <Value>Undergraduate</Value>
  </AcademicCareer>
  <UnitLevel Applicant="Y" Label="Unit level" Student="Y">
    <Level>Level 3</Level>
  </UnitLevel>
  <StaffList Applicant="Y" Label="Teaching staff" RoleLabel="Course Unit Role" Student="Y">
    <StaffMember>
      <Name>Yan Wang</Name>
      <Role>Unit coordinator</Role>
    </StaffMember>
  </StaffList>
  <OfferedBy Applicant="Y" Label="Offered by" Student="Y">
    <OrganisationList>
      <Organisation>
        <OrgName></OrgName>
      </Organisation>
    </OrganisationList>
    <GroupList>
      <Group>
        <GroupName></GroupName>
      </Group>
    </GroupList>
    <FheqLevels>
      <FheqLevel>
        <LevelNumber>1</LevelNumber>
        <LevelName>FHEQ level (Framework for Higher Education Qualifications) ' Last part of a Bachelors ' </LevelName>
      </FheqLevel>
    </FheqLevels>
    <Ects>
      <MaxUnits>European Credit Transfer &amp; Accumulation System Rating :   10.0</MaxUnits>
    </Ects>
  </OfferedBy>
  <MarketingOverview Applicant="Y" Label="Marketing Course unit overview" Student="">
    <Content>&lt;p&gt;The availability of text data has increased exponentially in recent years, alongside a growing demand for its analysis. This course introduces students to the quantitative analysis of text from a social science perspective, with a wide coverage of applications in economics, sociology &amp;amp; communication, and political science. The course adopts an applied approach: while theoretical aspects will be addressed, the primary objective is to equip students with the skills to formulate research questions that can be explored through text data and to understand the methodologies required to answer them. To this end, we begin by explaining how text can be conceptualized and modelled quantitatively, examining methods for comparing textual data. Following this, we delve into both supervised and unsupervised techniques in considerable depth, before addressing several specialized topics pertinent to social science research. Ultimately, the course aims to enable students to undertake their own research projects using text as data, providing a foundation for more advanced and technical investigations.&lt;br/&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </MarketingOverview>
  <UnitOverview Applicant="" Label="Course unit overview" Student="Y">
    <Content>&lt;p&gt;The availability of text data has increased exponentially in recent years, alongside a growing demand for its analysis. This course introduces students to the quantitative analysis of text from a social science perspective, with a wide coverage of applications in economics, sociology &amp;amp; communication, and political science. The course adopts an applied approach: while theoretical aspects will be addressed, the primary objective is to equip students with the skills to formulate research questions that can be explored through text data and to understand the methodologies required to answer them. To this end, we begin by explaining how text can be conceptualized and modelled quantitatively, examining methods for comparing textual data. Following this, we delve into both supervised and unsupervised techniques in considerable depth, before addressing several specialized topics pertinent to social science research. Ultimately, the course aims to enable students to undertake their own research projects using text as data, providing a foundation for more advanced and technical investigations.&lt;br/&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </UnitOverview>
  <Aims Applicant="Y" Label="Aims" Student="Y">
    <Content>&lt;p&gt;The availability of text data has increased exponentially in recent years, alongside a growing demand for its analysis. This course introduces students to the quantitative analysis of text from a social science perspective, with a wide coverage of applications in economics, sociology &amp;amp; communication, and political science. The course adopts an applied approach: while theoretical aspects will be addressed, the primary objective is to equip students with the skills to formulate research questions that can be explored through text data and to understand the methodologies required to answer them. To this end, we begin by explaining how text can be conceptualized and modelled quantitatively, examining methods for comparing textual data. Following this, we delve into both supervised and unsupervised techniques in considerable depth, before addressing several specialized topics pertinent to social science research. Ultimately, the course aims to enable students to undertake their own research projects using text as data, providing a foundation for more advanced and technical investigations.&lt;br/&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </Aims>
  <LearningOutcomes Applicant="Y" Label="Learning outcomes" Student="Y">
    <Content>&lt;p class="MsoBodyText" style="margin:11.3pt 29.9pt .0001pt 10.8pt;"&gt;&lt;span lang="EN-US"&gt;The&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;primary objective&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;of&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;this&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;course&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;is&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;to&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;familiarize&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;students&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;with&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;machine learning&lt;/span&gt;&lt;span style="letter-spacing:-.15pt;" lang="EN-US"&gt;&amp;nbsp;&lt;/span&gt;&lt;span lang="EN-US"&gt;methods and contemporary quantitative text analysis techniques, equipping them with the skills needed to apply these statistical methods in their own research. In pursuit of this objective, students will also engage with foundational concepts in machine learning and statistics, cultivating skills that are applicable to a broad range of data and inference challenges. Additionally, students will have the opportunity to enhance their programming competencies and develop an original research project.&lt;/span&gt;&lt;p&gt;&lt;/p&gt;&lt;/p&gt;</Content>
  </LearningOutcomes>
  <Knowledge Applicant="Y" Label="Knowledge and understanding" Student="Y">
    <Content>&lt;p&gt;• Demonstrate a theoretical understanding of content analysis approaches and machine learning techniques&lt;/p&gt;</Content>
  </Knowledge>
  <IntellectualSkills Applicant="Y" Label="Intellectual skills" Student="Y">
    <Content>&lt;p&gt;• Visualise, describe, and critically assess quantitative text analysis in R/Python, utilizing advanced methods&lt;/p&gt;</Content>
  </IntellectualSkills>
  <PracticalSkills Applicant="Y" Label="Practical skills" Student="Y">
    <Content>&lt;p&gt;• Produce reports for academic and non-academic audiences&lt;/p&gt;</Content>
  </PracticalSkills>
  <TransferableSkills Applicant="Y" Label="Transferable skills and personal qualities" Student="Y">
    <Content>&lt;p&gt;• Design and execute small-scale projects applying machine learning to social science research questions using text data&lt;/p&gt;</Content>
  </TransferableSkills>
  <EmployabilitySkillsList Applicant="Y" Label="Employability skills" Student="Y">
    <Skill>
      <SkillId></SkillId>
      <SkillDescription></SkillDescription>
    </Skill>
  </EmployabilitySkillsList>
  <Syllabus Applicant="Y" Label="Syllabus" Student="Y">
    <Content>&lt;p&gt;Lecture Schedule&lt;br/&gt;(10 sessions of 2-hour lectures and weekly 1-hour computer lab sessions)&lt;/p&gt;&lt;p&gt;1. Introduction to Quantitative Text Analysis&lt;br/&gt;Overview of the field, its applications in social sciences, and fundamental principles of text as data.&lt;/p&gt;&lt;p&gt;2. Descriptive Statistical Methods for Text Analysis&lt;br/&gt;Exploration of foundational descriptive statistics in text analysis, focusing on word frequency, term-document matrices, and other basic text preprocessing and summarization techniques.&lt;/p&gt;&lt;p&gt;3. Supervised Techniques with Text Data I&lt;br/&gt;Dictionary-based approaches, including sentiment analysis and the application of tools such as LIWC and other content dictionaries.&lt;/p&gt;&lt;p&gt;4. Supervised Techniques with Text Data II&lt;br/&gt;Document classification, including precision and recall as evaluation metrics, the role of crowdsourcing in supervised learning, and comparisons of various commonly used classifiers.&lt;/p&gt;&lt;p&gt;5. Transition from Supervised to Unsupervised Techniques&lt;br/&gt;Introduction to machine learning fundamentals, covering support vector machines, k- nearest neighbours, random forests, tree-based methods, and ensemble models.&lt;/p&gt;&lt;p&gt;6. Unsupervised Techniques with Text Data I&lt;br/&gt;Basics of unsupervised learning, with a focus on dimensionality reduction methods, including principal component analysis and singular value decomposition.&lt;/p&gt;&lt;p&gt;7. Unsupervised Techniques with Text Data II&lt;br/&gt;Clustering methods for document classification, scaling techniques, and various topic modelling approaches (e.g., Latent Dirichlet Allocation, Structural Topic Modelling, and BERT-based models).&lt;/p&gt;&lt;p&gt;8. Word Embeddings&lt;br/&gt;Examination of word embeddings for semantic analysis, covering methods such as Word2Vec, GloVe, and embeddings derived from language models.&lt;br/&gt;&amp;nbsp;&lt;/p&gt;&lt;p&gt;9. Neural Network-Based Models&lt;br/&gt;Introduction to neural networks for text analysis, with a focus on recurrent neural networks, convolutional neural networks, and transformer architectures.&lt;/p&gt;&lt;p&gt;10. Advanced Applications of Large Language Models (LLMs)&lt;br/&gt;Exploration of recent developments in LLMs, with an emphasis on their applications, limitations, and ethical considerations in text analysis.&lt;/p&gt;&lt;p&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </Syllabus>
  <TeachingMethods Applicant="Y" Label="Teaching and learning methods" Student="Y">
    <Content>&lt;p&gt;Description of T&amp;amp;L Methods&lt;/p&gt;&lt;p&gt;Instruction will be conducted over a 10-week period, with each week comprising two one- hour lecture sessions. Additionally, students will engage in a weekly one-hour computer lab session to apply theoretical concepts through hands-on exercises&lt;/p&gt;&lt;p&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </TeachingMethods>
  <AssessmentMethods Applicant="Y" Label="Assessment methods" Student="Y">
    <IntroText> </IntroText>
    <Method>
      <MethodId>3</MethodId>
      <MethodName>Report</MethodName>
      <MethodWeight>100%</MethodWeight>
    </Method>
    <OtherDescription>&lt;p&gt;Formative Assessment (Assignments) – modelling and coding of text data in a single RMarkdown PDF/HTML document of both answers and code.&lt;/p&gt;&lt;p&gt;Final paper: final data analysis report summarizing key findings from quantitative text analysis, including visualization.&lt;br/&gt;Report may be substantive / technical in nature. (2,000 words (including code, tables and figures): 100%)&lt;/p&gt;&lt;p&gt;&amp;nbsp;&lt;/p&gt;</OtherDescription>
  </AssessmentMethods>
  <FeedbackMethods Applicant="Y" Label="Feedback methods" Student="Y">
    <Content></Content>
  </FeedbackMethods>
  <RequirementsList Applicant="Y" Label="Pre/co-requisites" Student="Y">
    <Requirement>
      <UnitCode></UnitCode>
      <UnitTitle></UnitTitle>
      <RequirementType></RequirementType>
      <Description></Description>
    </Requirement>
  </RequirementsList>
  <AcademicPrograms Applicant="Y" Label="Academic programmes" Student="Y">
    <AcademicProgram>
      <Program></Program>
      <Plan></Plan>
      <Level></Level>
      <Requirement></Requirement>
    </AcademicProgram>
  </AcademicPrograms>
  <FreeChoice Applicant="Y" Label="Available as a free choice unit?" Student="Y">
    <Content></Content>
  </FreeChoice>
  <Accreditation Applicant="Y" Label="Accreditation" Student="Y">
    <Content></Content>
  </Accreditation>
  <RecommendedReading Applicant="Y" Label="Recommended reading" Student="Y">
    <Content>&lt;p&gt;Core Textbook:&lt;br/&gt;Grimmer, Justin, Margaret E. Roberts and Brandon M. Stewart (2022). Text as Data: A New Framework for Machine Learning and the Social Sciences. Princeton University Press, Princeton, NJ. This textbook is a recent survey of quantitative text analysis as used in the social sciences.&lt;/p&gt;&lt;p&gt;Supplementary Texts:&lt;br/&gt;• Jurafsky, Daniel and James H. Martin (2024). Speech and Language Processing: An Introduction to Natural Language Processing, Computational Linguistics, and Speech Recognition with Language Models. 3rd edition. Online manuscript released August 20, 2024. Available at https://web.stanford.edu/~jurafsky/slp3. This is a great reference book for the more technical aspects of quantitative text analysis.&lt;br/&gt;• Van Atteveldt, W., Trilling, D., &amp;amp; Calderon, C. A. (2022). Computational analysis of communication. John Wiley &amp;amp; Sons. Available at https://cssbook.net/ with codes and practices.&lt;/p&gt;&lt;p&gt;&amp;nbsp;&lt;/p&gt;</Content>
  </RecommendedReading>
  <StudyHours Applicant="Y" Label="Study hours" Student="Y">
    <IntroText> </IntroText>
    <ScheduledHours Applicant="Y" Label="Scheduled activity hours" Student="Y">
      <ActivityHours>
        <ActivityType>Lectures</ActivityType>
        <Hours>20</Hours>
      </ActivityHours>
      <ActivityHours>
        <ActivityType>Tutorials</ActivityType>
        <Hours>10</Hours>
      </ActivityHours>
    </ScheduledHours>
    <PlacementHours Applicant="Y" Label="Placement hours" Student="Y">
      <ActivityHours>
        <ActivityType></ActivityType>
        <Hours></Hours>
      </ActivityHours>
    </PlacementHours>
    <TotalHours Applicant="Y" Label="Independent study hours" Student="Y">
      <Hours>170</Hours>
    </TotalHours>
  </StudyHours>
  <Notes Applicant="Y" Label="Additional notes" Student="Y">
    <Content></Content>
  </Notes>
</CourseUnit>
