#set document(title: "15.6 Summary", author: "OpenStax / XYZ Homework") #set page(width: 8.5in, height: auto, margin: 1in) #import "@preview/cetz:0.5.2" #set text(font: ("STIX Two Text", "Libertinus Serif", "New Computer Modern"), size: 10.5pt, lang: "en") #show math.equation: set text(font: ("STIX Two Math", "New Computer Modern Math")) #set par(justify: true, leading: 0.62em, spacing: 0.9em) #set enum(spacing: 1.1em) // room between list items so tall inline fractions don't collide #set list(spacing: 1.1em) #set table(stroke: 0.5pt + rgb("#c7ccd3")) #let BLUE = rgb("#183B6F") // brand navy — section bars + example/solution labels (white on navy 11.09:1) #let ORANGE = rgb("#A94509") // brand primary-700 — AA-safe deep orange for TEXT (5.93:1 on white; raw brand #F37021 is 2.94:1 and must never carry text) #let RED = rgb("#DC2626") // brand error-600 #let GREEN = rgb("#059669") // brand success-600 (decoration only; small green text uses green-text #007942) #show heading.where(level: 1): it => block(width: 100%, above: 0pt, below: 16pt, fill: gradient.linear(BLUE, rgb("#2C5AA0")), inset: (x: 14pt, y: 12pt), radius: 3pt, text(fill: white, weight: "bold", size: 19pt, it.body)) #show heading.where(level: 2): it => block(width: 100%, above: 18pt, below: 10pt, fill: BLUE, inset: (x: 10pt, y: 6pt), radius: 2pt, text(fill: white, weight: "bold", size: 12pt, it.body)) #show heading.where(level: 3): it => text(fill: ORANGE, weight: "bold", size: 12.5pt, it.body) #show heading.where(level: 4): it => text(fill: BLUE, weight: "bold", size: 10.5pt, it.body) #let examplebox(label, title, body) = block(width: 100%, breakable: true, fill: rgb("#EFF1F5"), stroke: 0.5pt + rgb("#CFDDF0"), radius: 4pt, inset: 10pt, above: 12pt, below: 12pt)[ #block(below: 6pt)[#box(fill: BLUE, inset: (x: 6pt, y: 2pt), radius: 2pt, text(fill: white, weight: "bold", size: 8.5pt, label)) #h(0.4em) #strong[#title]] #body] // rail = decorative left rule (raw brand token); labelcolor = AA-safe label text shade #let notebox(label, rail, labelcolor, tint, body) = block(width: 100%, breakable: true, fill: tint, stroke: (left: 3pt + rail), inset: (left: 10pt, rest: 8pt), radius: (right: 4pt), above: 11pt, below: 11pt)[ #text(fill: labelcolor, weight: "bold", size: 7.5pt, tracking: 0.5pt)[#upper(label)] #linebreak() #body] #let solutionbox(body) = block(above: 4pt, below: 8pt)[ #text(fill: BLUE, weight: "bold", size: 8.5pt)[Solution] #linebreak() #body] #let figph(msg) = block(width: 100%, height: 60pt, fill: rgb("#f6f7f9"), stroke: (paint: rgb("#c7ccd3"), dash: "dashed"), radius: 4pt, inset: 10pt)[ #align(center + horizon, text(fill: rgb("#889"), style: "italic", size: 9pt, msg))] // Standardize inlined figure sizes: measure the natural CeTZ canvas, then scale to a // consistent envelope (aspect-aware; see build_typst.py FIG_* constants). Unlike the // print preamble, dimensions are FLOORED: in an editor a user can trim a figure to a // degenerate 1-D shape (a bare line), and w/h or tw/w would then divide by zero. #let _STD_W = 3.5 #let _WIDE_W = 5.6 #let _MAX_H = 3.4 #let _ASPECT_WIDE = 2.2 #let _UPSCALE_MAX = 1.15 #let stdfig(body) = context { let m = measure(body) let w = calc.max(m.width / 1in, 0.01) let h = calc.max(m.height / 1in, 0.01) let tw = if w / h > _ASPECT_WIDE { _WIDE_W } else { _STD_W } let s = calc.min(tw / w, _MAX_H / h, _UPSCALE_MAX) align(center, box(scale(x: s * 100%, y: s * 100%, reflow: true, body))) } #show figure: set block(breakable: false) #set figure(gap: 8pt) #show figure.caption: set text(size: 8.5pt, fill: rgb("#555")) == 15.6#h(0.6em)Summary Highlights from this chapter include: - Data science is a multidisciplinary field that combines collection, processing, and analysis of large volumes of data to extract insights and drive informed decision-making. - The data science life cycle is the framework followed by data scientists to complete a data science project. - The data science life cycle includes 1) data acquisition, 2) data exploration, 3) data analysis, and 4) reporting. - Google Colaboratory is a cloud-based Jupyter Notebook environment that allows programmers to write, run, and share Python code online. - NumPy (Numerical Python) is a Python library that provides support for efficient numerical operations on large, multi-dimensional arrays and serves as a fundamental building block for data analysis in Python. - NumPy implements an ndarray object that allows the creation of multi-dimensional arrays of homogeneous data types and efficient data processing. - NumPy provides functionalities for mathematical operations, array manipulation, and linear algebra operations. - Pandas is an open-source Python library used for data cleaning, processing, and analysis. - Pandas provides Series and DataFrame data structures, data processing functionality, and integration with other libraries. - Exploratory Data Analysis (EDA) is the task of analyzing data to gain insights, identify patterns, and understand the underlying structure of the data. - A feature is an individual variable or attribute that is calculated from raw data in a dataset. - Data indexing can be used to select and access specific rows and columns. - Data slicing refers to selecting a subset of rows and/or columns from a DataFrame. - Data filtering involves selecting rows or columns based on certain conditions. - Missing values in a dataset can occur when data are not available or were not recorded properly. - Data visualization has a crucial role in data science for understanding the data. - Different types of visualizations include bar plot, line plot, scatter plot, histogram plot, and box plot. - Several Python data visualization libraries exist that offer a range of capabilities and features to create different plot types. These libraries include Matplotlib, Seaborn, and Plotly. - The conventional aliases for importing NumPy, Pandas, and Matplotlib.pyplot are np, pd, and plt, respectively. At this point, you should be able to write programs to create data structures to store different datasets and explore and visualize datasets. #figure(table( columns: 2, align: left, inset: 6pt, table.header([Function], [Description]), [np.array()], [Creates an ndarray from a list or tuple.], [np.zeros()], [Creates an array of zeros.], [np.ones()], [Creates an array of ones.], [np.random.rand(n, m)], [Creates an array of random numbers with n rows and m columns], [np.genfromtxt('data.csv', delimiter=',')], [Creates an array from a CSV file.], [pd.DataFrame()], [Creates a DataFrame from a list, dictionary, or an array.], [pd.read\_csv()], [Creates a DataFrame from a CSV file.], [df.head()], [Returns the first few rows of a DataFrame.], [df.tail()], [Returns the last few rows of a DataFrame.], [df.info()], [Provides a summary of the DataFrame, including the column names, data types, and the number of non-Null values.], [df.describe()], [Generates the column count, mean, standard deviation, minimum, maximum, and quartiles.], [df.value\_counts()], [Counts the occurrences of unique values in a column and presents them in descending order.], [df.unique()], [Returns an array of unique values in a column.], [loc\[\]], [Allows for accessing data in a DataFrame using row/column labels.], [iloc\[\]], [Allows for accessing data in a DataFrame using row/column integer-based indexes.], [df\[condition\]], [Selects only the rows that meet the given condition.], [df.loc\[start\_row:end\_row, start\_column:end\_column\]], [Slices using label ranges.], [df.loc\[\[label1, label2, ...\], :\]], [Slices rows that are in the list \[label1, label2, ...\].], [df.isnull()], [Returns a Boolean array with Boolean values representing whether each entry has been Null.], [fillna()], [Replaces Null values.], [dropna()], [Removes all rows containing a Null value.], [plt.bar(x, height)], [Takes in two inputs, x and height, and plots bars for each x value with the height given in the height variable.], [plt.plot(x, y)], [Takes in two inputs, x and y, and plots lines connecting pairs of (x, y) values.], [plt.scatter(x, y)], [Takes in two inputs, x and y, and plots points representing (x, y) pairs.], [plt.hist(x)], [Takes in one input, x, and plots a histogram of values in x to show distribution or trend.], [plt.boxplot(x)], [Takes in one input, x, and represents minimum, maximum, first, second, and third quartiles, as well as outliers in x.], ))