Exploratory Analysis

Market Baseline: What Does the Data Analyst -> Data Scientist Market Look Like in NAICS 5182?

This page answers our Step 2 question: what does the market currently look like for our selected career pathway inside our selected industry? All charts below are generated live from data/processed/met_text_panel.csv (the course Jobs_2026 job files) – see Data Preparation for how this dataset was built and cleaned.

Chart theme: every chart on this page is interactive (built with Plotly) and uses one shared color palette. Hover over a bar to see exact values and sample sizes, drag to zoom, and double-click to reset.

import pandas as pd
import plotly.express as px
import plotly.graph_objects as go

# Consistent color theme used across every chart on this page
PALETTE = ["#457b9d", "#2a9d8f", "#e9c46a", "#e76f51", "#6d597a", "#f4a261"]
GRAY = "#b8bec6"
# the same work arrangement always gets the same color
ARRANGEMENT_COLORS = {"remote": PALETTE[0], "hybrid": PALETTE[1], "onsite": PALETTE[2], "unknown": PALETTE[3]}
ARRANGEMENT_LABELS = {"remote": "Remote", "hybrid": "Hybrid", "onsite": "Onsite", "unknown": "Not stated"}

raw = pd.read_csv("data/processed/met_text_panel.csv")


def highest_degree(row):
    # the highest degree level the posting text names, whatever the wording; None when there is no text
    if not row["has_text"]:
        return None
    for col, label in [("degree_PhD", "PhD"), ("degree_Master", "Master's"), ("degree_Bachelor", "Bachelor's")]:
        if row[col] in ("required", "preferred", "mentioned"):
            return label
    return "Not specified"


df = pd.DataFrame({
    "job_id": raw["job_id"],
    "title": raw["title"],
    "state_clean": raw["state"].fillna("").replace({"": "Unknown"}),
    "company_name_clean": raw["company_name_clean"],
    "remote_status": raw["remote_status"].fillna("Unknown").str.lower(),
    "salary_min_annual": raw["salary_min_annual"],
    "salary_max_annual": raw["salary_max_annual"],
    "experience_min_years": raw["experience_min_years"],
    "has_text": raw["has_text"],
})
df["education_level"] = raw.apply(highest_degree, axis=1)
TOTAL = len(df)
print(f"Total postings in baseline: {TOTAL}")

SCOPE = "Analyst-to-Scientist pathway, NAICS 5182, US"
ROLE_LABELS = {"ml engineer": "ML Engineer", "bi analyst": "BI Analyst"}


def role_label(key):
    return ROLE_LABELS.get(key, key.title())


def style(fig, title, subtitle, height=480, legend=False):
    # chart name, then a gray line saying which postings the chart covers and how many
    fig.update_layout(
        title=dict(text=f"<b>{title}</b><br><span style='font-size:12px;color:gray'>{subtitle}</span>",
                   x=0, xanchor="left"),
        template="simple_white", height=height, showlegend=legend,
        font=dict(family="Segoe UI, Helvetica, Arial, sans-serif", size=13),
        margin=dict(l=10, r=20, t=95, b=50),
        hoverlabel=dict(bgcolor="white", font_size=13),
        legend=dict(orientation="h", x=1, xanchor="right", y=1.02, yanchor="bottom"),
    )
    fig.update_xaxes(gridcolor="#eceff3")
    fig.update_yaxes(gridcolor="#eceff3")
    return fig


def show(fig):
    fig.show(config={"displaylogo": False, "responsive": True, "modeBarButtonsToRemove": ["lasso2d", "select2d"]})


def hbar(labels, values, xtitle, title, subtitle, texts=None, custom=None, hover=None, color=None, money=False, height=480):
    # horizontal bar chart; labels are listed bottom to top, so pass them sorted ascending
    fig = go.Figure(go.Bar(
        x=list(values), y=list(labels), orientation="h", marker_color=color or PALETTE[0],
        text=list(texts) if texts is not None else list(values), textposition="outside", cliponaxis=False,
        customdata=None if custom is None else list(custom),
        hovertemplate=hover or "<b>%{y}</b><br>%{x:,.0f}<extra></extra>"))
    fig.update_xaxes(title=xtitle, range=[0, max(values) * 1.18])
    if money:
        fig.update_xaxes(tickprefix="$", tickformat=",.0f")
    return style(fig, title, subtitle, height)


def salary_bars(stats, keys, labels, colors, title, subtitle):
    # average salary as bars, median as a diamond, sample size under each label
    xs = [f"{labels[k]}<br>(n={int(stats.loc[k, 'count'])})" for k in keys]
    label_colors = ["#1d2733" if colors[k] in (PALETTE[2], GRAY) else "white" for k in keys]
    fig = go.Figure()
    fig.add_trace(go.Bar(
        x=xs, y=[stats.loc[k, "mean"] for k in keys], name="Average", marker_color=[colors[k] for k in keys],
        text=[f"${stats.loc[k, 'mean']:,.0f}" for k in keys], textposition="inside", insidetextanchor="start",
        textfont=dict(color=label_colors, size=14),
        customdata=[[stats.loc[k, "median"]] for k in keys],
        hovertemplate="<b>%{x}</b><br>Average: $%{y:,.0f}<br>Median: $%{customdata[0]:,.0f}<extra></extra>"))
    fig.add_trace(go.Scatter(
        x=xs, y=[stats.loc[k, "median"] for k in keys], mode="markers", name="Median",
        marker=dict(symbol="diamond", size=12, color="black", line=dict(color="white", width=1)),
        hovertemplate="Median: $%{y:,.0f}<extra></extra>"))
    fig.update_yaxes(title="Estimated annual salary (USD)", tickprefix="$", tickformat=",.0f",
                     range=[0, stats["mean"].max() * 1.15])
    style(fig, title, subtitle, height=520, legend=True)
    fig.update_layout(legend=dict(orientation="h", x=0.5, xanchor="center", y=-0.2, yanchor="top"),
                      margin=dict(b=110))
    return fig
Total postings in baseline: 765

Job Volume

Job Volume by Role Type

vol = df["role_bucket"].value_counts().sort_values()

fig = hbar([role_label(k) for k in vol.index], vol.values, "Number of postings",
           "Job Volume by Role Type", f"{SCOPE} | {TOTAL} postings",
           custom=vol.values / TOTAL * 100,
           hover="<b>%{y}</b><br>%{x} postings (%{customdata:.0f}% of all)<extra></extra>")
show(fig)

Interpretation: Data Engineer roles dominate this pool (273 of 765 postings), followed by Data Scientist (204), Data Analyst (81), and Machine Learning Engineer (79, plus 21 titled ML Engineer). Pure “Data Analyst” titles are only about 11% of postings, which suggests that within NAICS 5182 the market skews toward more senior and technical data roles than entry-level analyst positions.

Salary

Salary Distribution

sal_df = df.dropna(subset=["salary_mid"])

fig = px.histogram(sal_df, x="salary_mid", nbins=20, marginal="box",
                   labels={"salary_mid": "Estimated annual salary (USD)"})
fig.update_traces(marker_color=PALETTE[0], selector=dict(type="histogram"))
fig.update_traces(marker_color=PALETTE[0], selector=dict(type="box"))
fig.update_traces(marker_line=dict(color="white", width=1), selector=dict(type="histogram"))
fig.update_traces(hovertemplate="Salary: %{x}<br>Postings: %{y}<extra></extra>", selector=dict(type="histogram"))
fig.update_yaxes(title_text="Number of postings", row=1, col=1)
fig.update_xaxes(tickprefix="$", tickformat=",.0f")
style(fig, "Salary Distribution", f"{SCOPE} | {len(sal_df)} postings with salary disclosed")
show(fig)

print(sal_df["salary_mid"].describe().round(0))
count       303.0
mean     195191.0
std       61408.0
min       50500.0
25%      154000.0
50%      188100.0
75%      227750.0
max      362500.0
Name: salary_mid, dtype: float64

Interpretation: About 40% of postings (303 of 765) disclose a salary. Among those, the middle half of postings pays between about $154K and $228K, the median is $188K, and a tail reaches about $363K, driven by senior Data Scientist and ML Engineer roles. “Estimated” here means the midpoint of the posted range – see Data Preparation for exactly how this was derived.

Average Salary by Role Type

role_sal = sal_df.groupby("role_bucket")["salary_mid"].agg(["mean", "count"])
role_sal = role_sal[role_sal["count"] >= 2].sort_values("mean")

fig = hbar([role_label(k) for k in role_sal.index], role_sal["mean"].round(0), "Average estimated annual salary (USD)",
           "Average Salary by Role Type", f"{SCOPE} | {len(sal_df)} salary-disclosed postings, roles with 2+",
           texts=[f"${v:,.0f}" for v in role_sal["mean"]], custom=role_sal["count"], money=True,
           hover="<b>%{y}</b><br>Average: $%{x:,.0f}<br>Postings with salary: %{customdata}<extra></extra>")
show(fig)

Interpretation: Machine Learning Engineer and Data Scientist roles command the highest average pay (about $224K to $234K for ML Engineer titles and $215K for Data Scientist), while Data Analyst and Business Analyst sit lowest (about $124K to $127K) – a roughly 1.8x pay gap between Data Scientist / ML and Data Analyst roles across this pathway. The move from Data Analyst to Data Scientist averages about $91K more per year, and role is the strongest pay factor in our Predictive Modeling page, so this is the most concrete number for prioritizing upskilling. Roles with fewer than 10 salary postings (for example, ML Engineer with 5) are shown but rest on small samples.

Top 10 Highest-Paying States

state_sal_df = sal_df[sal_df["state_clean"] != "Unknown"]
state_sal = state_sal_df.groupby("state_clean")["salary_mid"].agg(["mean", "count"])
state_sal = state_sal[state_sal["count"] >= 2].sort_values("mean", ascending=False).head(10).iloc[::-1]

fig = hbar(state_sal.index, state_sal["mean"].round(0), "Average estimated annual salary (USD)",
           "Top 10 Highest-Paying States",
           f"Roles pooled | {len(state_sal_df)} salary-disclosed postings, states with 2+",
           texts=[f"${v:,.0f}" for v in state_sal["mean"]], custom=state_sal["count"], money=True,
           hover="<b>%{y}</b><br>Average: $%{x:,.0f}<br>Postings with salary: %{customdata}<extra></extra>")
show(fig)

Interpretation: California ($225K) leads on average disclosed pay, followed by Massachusetts ($213K), New Jersey ($204K), Washington ($203K), and New York ($200K). California and New York also carry by far the most salary postings (85 and 76), so their averages are the most reliable; the other states in the chart rest on 3 to 12 salary postings. Because these averages pool all pathway roles, a state’s average also reflects its mix of higher- and lower-paying roles.

Location Patterns

loc = df[df["state_clean"] != "Unknown"]["state_clean"].value_counts().head(10).sort_values()

fig = hbar(loc.index, loc.values, "Number of postings", "Top 10 States by Job Volume",
           f"{SCOPE} | postings with a resolved US state",
           custom=loc.values / TOTAL * 100,
           hover="<b>%{y}</b><br>%{x} postings (%{customdata:.0f}% of all)<extra></extra>")
show(fig)

Interpretation: California and New York lead by a clear margin (141 and 113 postings, together 39% of postings with a resolved state), with Texas, Virginia, and Washington next – consistent with major tech/finance hubs and the Northern Virginia data-center corridor. 15% of postings (112 of 765) had no resolvable US state and are excluded from this chart rather than misattributed. Since California and New York combined account for a large share of postings, a job seeker unwilling to relocate to either state should expect a meaningfully smaller set of opportunities in this specific industry, even though remote options exist.

Experience Requirement Distribution

exp_df = df.dropna(subset=["experience_min_years"])

fig = go.Figure(go.Histogram(x=exp_df["experience_min_years"], xbins=dict(start=-0.5, end=15.5, size=1),
                             marker=dict(color=PALETTE[0], line=dict(color="white", width=1)),
                             hovertemplate="%{x:.0f} years<br>%{y} postings<extra></extra>"))
fig.update_xaxes(title="Minimum years of experience required", dtick=1)
fig.update_yaxes(title="Number of postings")
style(fig, "Experience Requirement Distribution", f"{SCOPE} | {len(exp_df)} of {TOTAL} postings state experience")
show(fig)

Interpretation: Where stated (206 of 765 postings, 27%), most postings cluster around 3-8 years of experience (73% of them), with a median of 5 years – this is a market that skews mid-to-senior level rather than entry-level, and only 17% ask for two years or fewer. That is a direct, actionable signal for a Data Analyst-pathway job seeker about how much experience they may need to build before this specific industry’s roles become attainable. For a candidate currently below the 3-year threshold, this suggests the realistic path into this industry may run through an adjacent, less experience-gated role first, rather than applying directly to NAICS 5182 postings.

Education Requirements

How we count. Here each posting is counted once, under the highest degree level its posting text names (PhD, then Master’s, then Bachelor’s), and a posting whose text names none counts as Not specified. The 195 postings without posting text are left out. How the posting words the degree (required or preferred) is not considered in this chart; the chart below separates it. Nothing is inferred.

Postings by Highest Degree Named

edu_order = ["Bachelor's", "Master's", "PhD", "Not specified"]
edu_colors = {"Bachelor's": PALETTE[0], "Master's": PALETTE[1], "PhD": PALETTE[2], "Not specified": GRAY}
txt_df = df[df["has_text"]]
edu_counts = txt_df["education_level"].value_counts().reindex(edu_order, fill_value=0)

fig = go.Figure(go.Bar(
    x=list(edu_counts.index), y=list(edu_counts.values), marker_color=[edu_colors[k] for k in edu_counts.index],
    text=[f"{v} ({v / len(txt_df):.0%})" for v in edu_counts.values], textposition="outside", cliponaxis=False,
    hovertemplate="<b>%{x}</b><br>%{y} postings<extra></extra>"))
fig.update_yaxes(title="Number of postings", range=[0, edu_counts.max() * 1.15])
style(fig, "Postings by Highest Degree Named in the Text",
      f"{SCOPE} | {len(txt_df)} postings with posting text, each counted once", height=450)
show(fig)

Interpretation: Of the 570 postings with text, 183 (32%) name a degree. A Bachelor’s is the highest degree named in 78 of them (14% of all 570), a PhD in 71 (12%), and a Master’s in 34 (6%), while 387 (68%) name no degree at all. Each posting is counted once under the highest level it names, so a posting that lists “Bachelor’s or Master’s” counts as Master’s. So in this industry slice, 34 postings have a Master’s as their highest named degree and 71 a PhD; the wording chart below shows how many of those actually state it as a requirement. Because most postings are silent on degrees, the counts show how employers phrase postings more than what the whole market demands.

Degree Wording in the Posting Text

The chart above counts a degree whenever the posting text names it. Here we separate how the posting words it, using the full text of the 570 postings that have it (see Data Preparation). A degree counts as required only when a sentence naming it also says require, minimum, must, or mandatory and has no preference word. A sentence with prefer, a plus, or nice to have counts as preferred, and a degree that is named with neither is named, no requirement wording. Nothing is inferred: a posting that names only a Master’s counts under Master’s alone, and a sentence such as “Master’s or PhD required” counts as required for both levels it names.

tp = pd.read_csv("data/processed/met_text_panel.csv")
tp = tp[tp["has_text"]]
levels = [("Bachelor", "Bachelor's"), ("Master", "Master's"), ("PhD", "PhD")]
wording = [("required", "Stated as required", PALETTE[1]),
           ("preferred", "Stated as preferred", PALETTE[2]),
           ("mentioned", "Named, no requirement wording", GRAY)]

fig = go.Figure()
for key, label, color in wording:
    counts = [int((tp["degree_" + col] == key).sum()) for col, _ in levels]
    fig.add_trace(go.Bar(
        x=[name for _, name in levels], y=counts, name=label, marker_color=color,
        text=[str(c) if c else "" for c in counts], textposition="inside",
        textfont=dict(color="#1d2733", size=13),
        hovertemplate="<b>%{x}</b><br>" + label + ": %{y} postings<extra></extra>"))
fig.update_layout(barmode="stack")
fig.update_yaxes(title="Number of postings", gridcolor="#eceff3")
style(fig, "How Postings Word Their Degree Requirement",
      f"MET Career Compass 2026 pathway postings with posting text | {len(tp)} postings", height=480, legend=True)
fig.update_layout(legend=dict(orientation="h", x=0.5, xanchor="center", y=-0.12, yanchor="top"), margin=dict(b=110))
show(fig)

Interpretation: Out of 570 postings with text, 11 state a Master’s as required and 4 state a PhD as required. Another 16 postings call a Master’s preferred and 13 call a PhD preferred, and 53 and 54 name them without any requirement wording. A Bachelor’s is stated as required in 25 postings, preferred in 10, and named without wording in 85. So a graduate degree is rarely a hard requirement in this sample: a PhD is more often a preference than a requirement, and nearly four in five postings (450 of 570) do not name a Bachelor’s at all. These counts are a floor, because a requirement can be written without these trigger words (for example, under a “Qualifications” heading in a separate sentence), and posting text is missing for about a quarter of the postings (195 of 765).

Remote Work Patterns

Remote vs Onsite vs Hybrid

order = ["remote", "hybrid", "onsite", "unknown"]
remote_counts = df["remote_status"].value_counts().reindex(order, fill_value=0)

fig = go.Figure(go.Bar(
    x=[ARRANGEMENT_LABELS[k] for k in order], y=list(remote_counts.values),
    marker_color=[ARRANGEMENT_COLORS[k] for k in order],
    text=[f"{v} ({v / TOTAL:.0%})" for v in remote_counts.values], textposition="outside", cliponaxis=False,
    hovertemplate="<b>%{x}</b><br>%{y} postings<extra></extra>"))
fig.update_yaxes(title="Number of postings", range=[0, remote_counts.max() * 1.15])
style(fig, "Remote vs Onsite vs Hybrid vs Not Stated", f"{SCOPE} | {TOTAL} postings", height=450)
show(fig)

Interpretation: The majority of postings (77%) don’t explicitly state a remote/onsite policy at all – itself a notable market finding, not a data gap. Among the 179 postings that do specify, remote (78) and hybrid (75) are about equally common, and onsite is much rarer (26).

Average Salary by Work Arrangement

order = ["hybrid", "remote", "onsite", "unknown"]
rem_sal = sal_df.groupby("remote_status")["salary_mid"].agg(["mean", "median", "count"]).reindex(order)

fig = salary_bars(rem_sal, order, ARRANGEMENT_LABELS, ARRANGEMENT_COLORS,
                  "Average Salary by Work Arrangement", f"{SCOPE} | {len(sal_df)} salary-disclosed postings")
show(fig)

Interpretation: Among postings that state a work arrangement, hybrid roles have the highest average disclosed salary ($193K), ahead of remote ($184K, median $200K), with onsite far behind ($136K), but the onsite group has only 10 postings, so that gap should be read with caution. Postings that do not state an arrangement average $199K (n=238); that group is the largest and mixes many roles and employers, so it says more about who omits the policy than about pay for any one arrangement. The Predictive Modeling page tests these differences after accounting for role and state.

Top Employers

emp_df = df[~df["company_name_clean"].str.startswith("Unknown")]
top_emp = emp_df["company_name_clean"].value_counts().head(10).sort_values()

fig = hbar(top_emp.index, top_emp.values, "Number of postings", "Top 10 Employers",
           f"{SCOPE} | excludes unattributable listings")
show(fig)

Interpretation: Amazon is the single largest poster in this pool (60 postings), consistent with AWS’s scale within the Computing Infrastructure Providers / Data Processing industry. Federal and defense contractors (Accenture Federal Services, Peraton, CACI) and large financial institutions (JPMorgan Chase, PNC) also appear prominently, reflecting how much cloud/data infrastructure work is now done at scale by non-“tech-native” companies. 4% of postings (27) have no attributable employer name (job-board listings) and are excluded from this ranking.

Summary

  • 765 postings make up our Data Analyst -> Data Scientist baseline within NAICS 5182 (course Jobs_2026 job files, 2026 postings)
  • Job volume skews toward Data Engineer and Data Scientist roles over entry-level Data Analyst titles
  • Disclosed salaries (40% of postings) have a $188K median, with a ~1.8x pay gap between Data Analyst and Data Scientist / ML Engineer averages
  • Experience requirements, where stated, mostly fall in the 3-8 year range – a mid-to-senior-skewed market
  • California and New York dominate location share, and California leads on average disclosed pay
  • Most postings (77%) don’t disclose remote/onsite policy; remote and hybrid are about equally common among those that do
  • Amazon, federal contractors, and large financial institutions are the top identifiable employers

These findings feed our Skill Gap Analysis and Predictive Modeling pages, which compare team skills with what the postings ask for and test which factors are associated with pay. The pay gap between roles is the reason to plan a move from Data Analyst toward Data Scientist / ML roles.