\documentclass[12pt]{article}
\usepackage{amsmath,amssymb}
\usepackage[letterpaper,width=6.5in,height=9in]{geometry}
\usepackage{harvard}
\bibliographystyle{econometrica}
\usepackage{enumerate}
\usepackage{color} 
\usepackage{verbatim} 
%\usepackage[mathscr]{euscript} 
\usepackage{mathrsfs} 
\usepackage{graphicx}
\usepackage{subcaption}
\usepackage{centernot}
\usepackage{accents}
\usepackage{epsfig}
\usepackage{latexsym}
\usepackage{amsmath,amssymb,bbm}
\usepackage{dcolumn}
\usepackage{enumitem} 
\usepackage{dutchcal}
 \usepackage{tikz}
\usepackage[T1]{fontenc} 
\usepackage{changepage}

%\usepackage{pslatex}
%\usepackage{pxfonts}
%\usepackage{eulervm}
%\setlength{\parindent}{0ex}
%\setlength{\parskip}{1.5ex}

\usepackage{setspace}
\usepackage{IEEEtrantools}
\onehalfspacing

%\varepsilonmberwithin{equation}{section}
\newtheorem{theorem}{Theorem}%[section]
\newtheorem{corollary}[theorem]{Corollary}
\newtheorem{proposition}[theorem]{Proposition}
\newtheorem{propositionaltl}{Proposition} 
\renewcommand{\thepropositionaltl}{\arabic{propositionaltl}$^\prime$}
\newtheorem{propositionalt}{Proposition} 
\renewcommand{\thepropositionalt}{\arabic{propositionalt}$^{\prime\prime}$}
\newtheorem{example}{Example}
\newtheorem{assumption}{Assumption}
\newtheorem{assumptionalt}{Assumption} 
\renewcommand{\theassumptionalt}{\arabic{assumptionalt}$^\prime$}
\newtheorem{assumptionm}{Assumption} 
\renewcommand{\theassumptionm}{M-\arabic{assumptionm}}
\newtheorem{assumptiona}{Assumption} 
\renewcommand{\theassumptiona}{\Alph{assumptiona}}
\newtheorem{lemma}{Lemma}
%\newtheorem{remark}{Remark}

\newenvironment{proof}{
  \noindent\textbf{Proof:}\ }{\hspace*{\fill}\\
  $\Box$}
\newenvironment{remark}{
  \noindent\textbf{Remark:}\ }{\hspace*{\fill}\medskip}
\newenvironment{remarks}{
  \noindent\textbf{Remarks:}\ }{\hspace*{\fill}\medskip}
\newcommand{\cvgw}[1]{\stackrel{#1}{\leadsto}}
\newcommand{\cvgp}{\stackrel{\rm p}{\longrightarrow}}
\newcommand{\cvgd}{\stackrel{\rm d}{\longrightarrow}}
\newcommand{\mathfrc}[1]{\text{\textfrc{#1}}}

\newcommand*{\Resize}[2]{\resizebox{#1}{!}{$#2$}}

%\DeclareFontFamily{OT1}{pzc}{}
%\DeclareFontShape{OT1}{pzc}{m}{it}{<-> s * [0.900] pzcmi7t}{}
%\DeclareMathAlphabet{\mathscr}{OT1}{pzc}{m}{it}

\begin{document} 




\title{Moment Inequalities for Multinomial Choice \\ with Fixed Effects}

\author{Ariel Pakes\\ Harvard University and NBER \\ \\  Jack
  Porter\\ University of Wisconsin} 


\date{\vskip1cm  October 17, 2020} 

\maketitle





\begin{abstract}
This paper proposes a new approach to identification of the 
semiparametric multinomial choice model with fixed effects.  
The framework employed is the semiparametric version of the traditional multinomial logit 
with fixed effects model (Chamberlain 1980).  This semiparametric multinomial choice model places no 
restrictions on either the joint distribution of the random utility disturbances across choices or their within group 
(or across time) correlations.  
We show that a novel within-group comparison leads to a set of conditional moment inequalities.  
Our main finding shows that the derived 
conditional moment inequalities yield the sharp identified set for the random utility covariate coefficients, while avoiding the incidental parameter problem.   Specializing this result to the binary choice case shows that 
Manski's (1987) conditional moment inequalities still lead to sharp bounds without restrictions on covariates.  





\end{abstract}

\clearpage 




\section{Introduction}



This paper characterizes identification of the 
 semiparametric 
multinomial choice model with fixed effects and a group (or panel)
structure.  A standard multinomial framework (\cite{mcfa:1974}) is employed with random utility  
that is additively separable between unobservables, which include a
disturbance and choice-specific fixed effects, and a linear covariate index.
The key semiparametric assumption, replacing the multinomial logit specification (Chamberlain 1980), 
is a familiar group stationary condition on the
disturbances.  This assumption places no restrictions on either 
the joint distribution of the disturbances across choices or the correlation of disturbances across time.  
%, so that arbitrary substitution patterns across choices are allowed.  
Under this specification, a novel within-group comparison leads to a 
 set of conditional moment inequalities, which are the basis for our result on 
 sharp partial identification.  
  
  
  
  Our main finding establishes sharp identification in the semiparametric multinomial choice model with fixed effects.  
  Under the group stationarity assumption alone, we find that our full set of 
derived conditional moment inequalities contains all of the model's potential identifying information in the sense that the bounds provided by 
these inequalities are sharp.  Sharpness is shown by a constructive proof.  In particular, given a distribution of observables and a parameter 
value in the identified set, we demonstrate that there exists a distribution of unobservables that can be combined with the parameter 
value to generate the given distribution of observables.  
%While fixed effects are not needed in this demonstration, allowing for correlation across time 
%in the constructed unobservable distribution is, in general, necessary.  



The semiparametric  model considered here does not place
parametric restrictions on the disturbance distribution.  The only
restriction on the disturbances is a group or time stationary assumption.  
Since the joint distribution of disturbances across choices is left 
unrestricted, the model contains no vestiges of
independence of irrelevant alternatives or limits on cross price
elasticities.  Within group or across time disturbance correlation is also 
left completely 
unrestricted in this specification.  The panel aspect of the model 
allows for an additive choice-specific fixed effect in the random utility 
specification.  The fixed effects are allowed to be arbitrarily correlated 
with the observed covariates.  We focus on the case with only two time periods (or group observations).  
The derived conditional moment inequalities are based on only within variation, so 
that the incidental parameter problem is fully circumvented in our identification results.  


  
 

The most closely related work is \citeasnoun{shi:shu:son:2018}, which obtains point identification using cyclic monotonicity in the semiparametric multinomial setup.  
Since our conditional moment inequalities provide sharp bounds, it is not surprising that we are able to 
show that the \citeasnoun{shi:shu:son:2018} conditional moment inequalities are implied by our 
conditional moment inequalities.  It follows that, under the additional conditions on covariates given in 
\citeasnoun{shi:shu:son:2018}, our conditional moment inequalities  also yield point identification.  

When we specialize our setup to the binary choice case, we find that our conditional moment inequalities 
match the weak version of \citeasnoun{mans:1987}'s conditional moment inequalities.  
It follows that these conditional moment inequalities yield sharp bounds even when point identification fails 
due to insufficient variation in the covariates.  
Further, we  establish that  \citeasnoun{mans:1987}'s  maximum score criterion can be derived as an aggregation of the conditional moments that make up our inequalities.  
We prove that the identified set determined by our conditional moment inequalities is exactly the set of parameters that maximize the maximum 
score criterion function.  This new result shows that the maximum score criterion can be used for (sharp) identification (and hence its sample counterpart can 
be used for estimation) even when point identification fails in the binary choice panel model. 





Multinomial discrete choice models are extensively used in almost
all fields that empirically analyze the determinants of agents'
choices.   
Applications have typically employed parametric forms of the multinomial
model. With panel data problems in mind,
\citeasnoun{cham:1980} uses an assumption of logistic disturbances to
provide a novel conditional likelihood method of identification and
estimation.  An alternative application is in the demand literature
where markets are the grouping device, the within group observations
are consumers, and the choice-specific fixed effects represent product
level unobservables (e.g.\ \cite{berr:levi:pake:1995}).  Markets are
also used as a grouping device when analyzing firm decision making
(e.g.\ entry decisions) with the market-specific fixed effect
representing unobserved determinants of the market's profitability
(e.g.\ \cite{pake:2014}). 

\citeasnoun{mans:1975} introduced a 
semiparametric, maximum score approach to point identification
and estimation for multinomial choice without choice-specific fixed
effects.  Assuming independent and identical distributions of the
unobservable components of the different choices, Manski uses
differences in the observable, parametric component of random utility
across choices for identification.  Using Manski's identification
approach, \citeasnoun{fox:2007} shows that exchangeability of the
unobservable component across choices is sufficient for identification, and
\citeasnoun{yan:2013} obtains the limiting distribution for a smoothed
version of the multinomial maximum score estimator.
\citeasnoun{lee:1995} provides an alternative semiparametric approach
to multinomial choice for models without choice-specific fixed effects
using an assumption of an i.i.d.\ distribution of disturbances across
agents. Rather than imposing conditions on the joint distribution of
the disturbances across choices our approach requires that the joint
distribution of the choice-specific unobservables does not differ across
observations in a group, but leaves the distribution of disturbances
across choices unrestricted. The different assumptions are likely to
be useful in different applications.  
  \citeasnoun{kahn:ouya:tame:2019}
develop an approach to identification that can be used in both static and dynamic semiparametric multinomial choice models.  
Using a nonparametric multinomial choice model with 
endogeneity for the California Health Insurance Exchange,   \citeasnoun{teba:torg:yang:2018} identify and estimate bounds on counterfactuals.   
\citeasnoun{gao:2018} develops estimation and identification in a non-separable 
version of the panel multinomial choice model without restricting the joint distribution of disturbances. 
%\\ 
%(TO BE ADDED: Shi-Shum-Song more careful comparison )









This work also continues a substantial literature that has focused on extending  nonlinear econometric models 
to allow for fixed effects while 
relaxing parametric distributional assumptions on disturbances.  \citeasnoun{mans:1987} applied his maximum score approach 
to  the binary choice model with fixed effects.  \citeasnoun{hono:1992} further developed \citename{powe:1986}'s (\citeyear{powe:1986}) 
trimmed least squares approach to estimate the censored regression model with fixed effects.  \citeasnoun{abre:1999} developed a 
new approach to estimation to allow for fixed effects in the transformation model, and further extended  \citename{han:1987}'s 
(\citeyear{han:1987}) generalized regression model to include fixed effects in  \citeasnoun{abre:2000}. \citeasnoun{ahn:ichi:powe:ruud:2017} 
develop an approach to identification and estimation of semiparametric index models that can also be used in various fixed effects cases.  
The multinomial choice setup considered in the current work presents
an additional complexity relative to the models in this previous
literature.  In particular, the multinomial choice model depends on
{\em multiple}  index functions of the covariates, where each index
function corresponds to a choice-specific random utility.  The main
insight of our identification strategy is that a comparison of the
multiple index functions for any two within-group observations has
observable implications on the relative likelihood of certain choice
outcomes.  














The paper is structured as follows.  Section~\ref{sec:2} 
sets up a semiparametric version the standard random utility model
for multinomial choice with fixed effects. 
We then introduce our main stochastic disturbance assumption 
and derive a set of conditional moment inequalities.  
In section~\ref{sec:sharp}, we show that the conditional moment inequalities 
provide sharp bounds on the parameters of interests. 
In section~\ref{sec:add}, we address point identification, binary choice, 
and estimation and inference.  
 Section~\ref{sec:con} concludes.  All proofs are in the Appendix.  









%%%%%%%%%%%%%%%
%
%   MOM INEQS
%
%%%%%%%%%%%%%%%




\section{Conditional Moment Inequalities for Multinomial Choice}
\label{sec:2}


%%%%%%%%%%%%%%%
%
%   SETUP
%
%%%%%%%%%%%%%%%

\subsection{Setup}
\label{sec:set} 

The data will be assumed to have a group/panel 
structure, where $i=1,\ldots,n$ indexes the groups and $t = 1,\ldots,T$ indexes observations
within a group.  There are a number of familiar multinomial choice
applications with this group structure. In panel data applications in
Labor and Public Finance, $i$ typically indexes individuals, and $t$
indexes time periods, though alternative groupings can also be
relevant (an example from the study of hospital choice has $i$
indexing illness categories and $t$ indexing the individuals in these
categories, see \citeasnoun{ho:pake:2014}).  In Industrial Organization and Marketing applications,
$i$ would typically index markets and $t$ would index either the
different consumers in those markets (in demand analysis) or the firms
that compete in them (in the analysis of a firm's choice of controls).


Observation $(i,t)$ faces a number of choices. Each choice $d$ has an
associated random utility, $U_{d,i,t}$, and the observed choice,
$y_{i,t}$, maximizes the random utility over choices.  
Suppose that  $d \in \{0, \ldots,\mathscr{D}\}$, so that the number of choices is 
 $\mathscr{D}+1$.    We consider the case of unordered response,
where the numbering associated with each choice is
arbitrary,\footnote{Inequalities for models with ordered responses are
  considered in \citeasnoun{pake:port:ho:ishi:2015}.}   and $2\le \mathscr{D}+1 < \infty$.
  

Given covariates $x_{d,i,t}$ for each choice $d$ associated with observation $(i,t)$, the random
utility for choices $d = 1, \ldots, \mathscr{D}$ takes the form
\begin{equation}
\label{eqn:u} 
U_{d,i,t} = x_{d,i,t}^\prime\theta_0 + \lambda_{d,i}+\varepsilon_{d,i,t},
\end{equation} 
where the term $\lambda_{d,i}$ denotes
choice-specific fixed effects which account for unobserved characteristics
of choice $d$ that do not vary across $t$.  No restrictions are placed on the correlation between covariates $x_{d,i,t}$ and choice-specific fixed effects $\lambda_{d,i}$, so these  
fixed effect terms generate a potential incidental parameter problem.  The term
$\varepsilon_{d,i,t}$ represents any remaining unobserved,
idiosyncratic determinants of the random utility.  
The covariates enter random utility through a  
linear index.  The additive separability between the covariate index $x_{d,i,t}^\prime\theta_0$ and the 
unobserved terms $\lambda_{d,i}+\varepsilon_{d,i,t}$ is critical to the results that follow.  However, 
the additive separability between the fixed effect $\lambda_{d,i}$ and disturbance 
$\varepsilon_{d,i,t}$ could be relaxed.  That is, $\lambda_{d,i}+\varepsilon_{d,i,t}$ 
could be replaced by a term of the form $f_d(\lambda_{d,i},\varepsilon_{d,i,t})$, where 
$f_d$ is an unknown nonlinear function for choice $d$.\footnote{Strictly speaking, 
under the  assumptions below, the    
fixed effect could be absorbed 
into the disturbance without loss of generality.}

The random utility for choice $d=0$ is normalized to be zero,
\[ 
U_{0,i,t} = 0
\]
so that the random utilities for all remaining choices are relative to choice $0$.  
Similarly, we normalize $x_{0,i,t}=0$ and $\lambda_{0,i} = \varepsilon_{0,i,t}=0$.  





The observed choice, $y_{i,t}$, for agent $(i,t)$ maximizes the random
utility $U_{d,i,t}$ over choices $d$. 
%\begin{equation} 
%\label{eqn:argmax} 
%y_{i,t} = \max \arg\!\max_d  U_{d,i,t}.
%\end{equation}
%where the $\arg\!\max$ function generates the set of choices that
%maximize random utility.  
When a single choice uniquely maximizes random utility, then 
 that choice  is 
the
observed choice for $(i,t)$.  
 We also allow for situations where there is a non-zero probability that two or more choices are maximize utility.  This situation 
 could occur if the distribution of $\varepsilon_{i,t}$ has mass points, as could be allowed under the flexible nonparametric assumptions 
 on disturbances here, or when there are set-valued regressors.\footnote{In the set-valued regressors case, 
 the researcher does not know the specific value of some regressors, but does observe a set that contains their values.  \citeasnoun{pake:port:2014} use the tools developed here to analyze this case. 
  Two familiar examples are when the regressor is: (i) income (or wealth) and all the econometrician knows is that the income of each observation lies in a particular interval; and (ii) the distance from home to a service (or retail) outlet when the home location is only observed as a zip code (with known geographic boundaries).}    
To fully specify the choice decision, we adopt a simple rule for resolving ties 
 among maximizing random utility choices.   If choices $d_1$ and $d_2$ both maximize 
 random utility ($U_{d_1,i,t} = U_{d_2,i,t} = \max_d U_{d,i,t}$) and $d_1 < d_2$, then assume that the larger of the choice numbers $d_2$ is the observed choice for $(i,t)$.  And, in general, 
if there are multiple utility maximizing
choices, then the observed outcome is assumed to be the largest 
 choice number among the utility maximizing choices.\footnote{Any non-stochastic rule for resolving 
 utility maximizing ties will suffice for the results that follow.}    
 






The setup thus far is a standard random utility formulation of
multinomial choice except that %, as in \citeasnoun{cham:1980}, 
a choice-specific group fixed effect is included.  It will be 
useful to establish notation for the mapping defined by this setup 
from the covariates, parameter $\theta_0$, fixed 
effects, and disturbances to the observed outcome $y_{i,t}$.  
Let $x_{i,t} = (x_{1,i,t}^\prime, \ldots, x_{\mathscr{D},i,t}^\prime)^\prime$, 
$\lambda_i = (\lambda_{1,i}, \ldots, \lambda_{\mathscr{D},i})$, 
$\varepsilon_{i,t} = (\varepsilon_{1,i.t}, \ldots, \varepsilon_{\mathscr{D},i,t})$, 
and $U_{i,t} = (U_{1,i,t}, \ldots, U_{\mathscr{D},i,t})$.  
When it is helpful to be explicit about the dependence of random utility on 
its components, we will use the notation $U_{d}(x_{i,t}, \theta_0, \lambda_i, \varepsilon_{i,t})$ 
to denote $U_{d,i,t} = x_{d,i,t}^\prime\theta_0 + \lambda_{d,i}+\varepsilon_{d,i,t}$.  
The observed outcome $y_{i,t}$ can also be written as a function of these same components:
$
y_{i,t} = \mathcal{y}(x_{i,t}, \lambda_i, \varepsilon_{i,t}, \theta_0) = \max \arg\!\max_d 
U_{d}(x_{i,t}, \theta_0, \lambda_i, \varepsilon_{i,t})$, where $\mathcal{y}$ is 
the mapping that represents the random utility formulation for multinomial choice 
given above.  


Our identification results will correspond to the case where $T$ is fixed, and, to further simplify 
the discussion, we will focus on the case where $T = 2$.  We will denote the two 
time periods or observations within each group by $s$ and $t$, rather than $1$ and $2$ 
to avoid confusion, especially in the variable subscripts, with the choices $d$ which are numbered $0, \ldots, \mathscr{D}$. 









\vspace{.7 cm}


The key stochastic assumption for this framework is  within-group/time homogeneity 
of the disturbances.  This assumption is sometimes called  {\it strict exogeneity} and is a common 
condition imposed in static panel data models (\cite{cher:fern:hahn:newe:2013}).\footnote{Mean independence and zero 
covariance forms of strict exogeneity also appear commonly in the literature, especially in linear panel data model cases.  Here, 
the stronger conditional independence form of strict exogeneity will be employed.}
\begin{assumption} 
\label{asm:orthog} 
~ 
\\ 
(a) $(x_{i,s}, x_{i,t}, \lambda_i, \varepsilon_{i,s}, \varepsilon_{i,t})$ is independently and identically distributed 
for $i = 1, \ldots, n$;\\ 
(b)
Given the conditioning set $(x_{i,s}, x_{i,t}, \lambda_i)$, the conditional distributions 
of $\varepsilon_{i,s}$ and  $\varepsilon_{i,t}$ are the same:
\[
 \varepsilon_{i,s} \big| x_{i,s}, x_{i,t}, \lambda_i  \ \sim \  \varepsilon_{i,t} \big| x_{i,s}, x_{i,t}, \lambda_i. 
\]
\end{assumption} 

\noindent 
The second part of the assumption mirrors the stochastic assumption made for 
panel data binary choice models in \citeasnoun{mans:1987} and for discrete choice in \citeasnoun{shi:shu:son:2018}.  
 No parametric
distributional restrictions are placed on the distribution of
$\varepsilon_{i,t}$.  
Note that $\varepsilon_{i,t}$ is, in general, a vector of individual choice disturbances, 
in contrast to the binary choice case.  
 Importantly, for a given time $t$, the marginal
distribution of these choice disturbances is allowed to vary arbitrarily across
choices ($d$), and there are no restrictions on joint behavior of these disturbances 
across choices.  As a result, neither independence of
irrelevant alternatives, nor any other limitation on the
substitutability of different choices induced by the covariance
structure of disturbances (such as the limited substitutability
property discussed in \cite{berr:pake:2007}) is a source of concern.  
This assumption also allows the disturbances for the different choices to be
freely correlated {\it across time}. 
Assumption~\ref{asm:orthog} nests both the familiar panel data
model with individual choice-specific fixed effects and i.i.d.
disturbances, a special case of which is Chamberlain's
(\citeyear{cham:1980}) conditional logit model, and many
differentiated product demand models for micro data
(e.g. \cite{berr:levi:pake:2004}). 
\begin{comment} 
{\color{blue} Jack; I am not sure but my presumption is that 1a is for convenience.  If it were not
true you would just have to restrict the distributions of each individual $i$ to have the properties you endow them with below when
you need those properties, but the whole distribution would not need to be same.  If that is true maybe we should say it,
as in the dynamic paper we don't need i.i.d. either.} \\ 
{\color{red}  NTS:  check role of iid-ness in proofs}
\end{comment} 






Assumption~\ref{asm:orthog} does restrict the relationship between the
disturbances and the covariates.  For instance, heteroskedasticity
would need to take a specific form where the heteroskedasticity in
$\varepsilon_{i,t}$ is the same as $\varepsilon_{i,s}$ even when
$x_{i,t} \ne x_{i,s}$.  For example, if the heteroskedasticity in both
$\varepsilon_{i,t}$ and $\varepsilon_{i,s}$ depended on $x_{i,t} +
x_{i,s}$, then Assumption~\ref{asm:orthog} would not be violated.    Of
course, the typical assumption of independence of disturbances and covariates across different
$s$ and $t$ would suffice to satisfy Assumption~\ref{asm:orthog}.


Assumption~\ref{asm:orthog}(b) means that ``within'' variation 
could be useful for identification.  
By restricting the conditional joint distribution of the disturbances
across the random utility choices to be the same for observations in
group $i$, Assumption~\ref{asm:orthog} enables us to learn about
relative response probabilities by comparing the {\it observable}
components of random utilities across $t$ for that group $i$.  This within-group 
comparison will not depend on the joint distribution of disturbances
across choices in any way.  




To simplify notation, below we eliminate the
group $i$ index with the understanding that all variables below are
associated with the same group unless otherwise indicated.












%%%%%%%%%%%%%%%
%
%   EG MOM INEQ
%
%%%%%%%%%%%%%%%

\subsection{Illustrative Moment Inequality} 
\label{sec:1mom} 


Given the random utility framework above along with
Assumption~\ref{asm:orthog}, we can derive a set of moment inequality
conditions that can be taken to data for inference on the parameter
$\theta_0$.  We begin with a single conditional moment inequality that
makes both the assumptions and logic underlying our conditional moment
inequality analysis transparent.  Following this derivation, we show
how an extension of this logic leads to a collection of
conditional moment inequalities.  



Our moment inequalities are based on a within comparison of choice probabilities for 
individual/group $i$ at times $s$ and $t$.  
We can express the conditional probability of observing choice $d$ at time $t$ through the 
corresponding region of the disturbance space. 
\begin{equation} 
\mathcal{E}_{d,t} = \{ \varepsilon_t:  \mathcal{y}(x_{t}, \lambda, \varepsilon_{t}, \theta_0) = d\} 
\label{eqn:dreg}
\end{equation} 

\begin{comment} 
\begin{eqnarray} 
\mathcal{E}_{d,t} &=& \{ \varepsilon_t:  \mathcal{y}(x_{t}, \lambda, \varepsilon_{t}, \theta_0) = d\} 
\nonumber 
\\[1ex]  
&= & \left\{ \varepsilon_t:   \varepsilon_{d,t} \,\ge \, \max_{c < d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}  
\label{eqn:reg}
\\ && 
\ \cap \ \left\{ \varepsilon_t:   \varepsilon_{d,t} \, >  \, \max_{c > d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}  
\nonumber 
\end{eqnarray} 
Given our rule for resolving utility maximizing ties, $\mathcal{E}_{d,t}$ could also be expressed as 
$ \left\{ \varepsilon_t:   \varepsilon_{d,t} \,\ge \, \max_{c < d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}$ 
$\ \cap \ \left\{ \varepsilon_t:   \varepsilon_{d,t} \, >  \, \max_{c > d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}$.  
\end{comment} 
And, given this definition, $\Pr(y_t = d | x_s, x_t,\lambda) = \Pr( \varepsilon_t \ \in \ \mathcal{E}_{d,t}\, | \, x_s, x_t,\lambda )$.  
%
%Since larger numbered choices are assigned random utility ties with smaller numbered choices, the region above 
%requires a strict inequality for choices $c >d$ and only a weak inequality for choices $c <d$.  
%Given this set definition, 
%\begin{equation} 
%\Pr(y_t = d | x_s, x_t,\lambda) = \Pr( \varepsilon_t \ \in \ \mathcal{E}_{d,t}\, | \, x_s, x_t,\lambda )
%\nonumber 
%\label{eqn:cprob}
%\end{equation} 
%This probability involves the difference of the covariate index functions %and the fixed effects 
%across the choices.  
%
To consider how variation in the covariates across time affects choice probabilities, it is useful to note the explicit dependence of the region $\mathcal{E}_{d,t}$ on the covariates, which follows from our rule for resolving utility maximizing ties:
\begin{eqnarray} 
\mathcal{E}_{d,t} &=&  \left\{ \varepsilon_t:   \varepsilon_{d,t} \,\ge \, \max_{c < d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}  
\label{eqn:reg}
\\ && 
\ \cap \ \left\{ \varepsilon_t:   \varepsilon_{d,t} \, >  \, \max_{c > d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}.  
\nonumber 
\end{eqnarray} 
%
%
%We consider the  variation in the covariate index function over choices and how changes in 
%this variation over time affect the choice probabilities.   In particular, if 
We compare the time $t$ regions $\mathcal{E}_{0,t}, 
\ldots, \mathcal{E}_{\mathscr{D},t}$ to the analogous regions at time $s$, $\mathcal{E}_{0,s}, 
\ldots, \mathcal{E}_{\mathscr{D},s}$.  From this comparison, we will be able to show that for one of the $\mathscr{D}+1$ choices, the region at time $s$ contains 
the corresponding region at time $t$.  Moreover, the choice with this property is determined completely by the covariate indices.   Assumption~\ref{asm:orthog}   then implies that the corresponding choice probability at time $s$ will be at least as large as  the choice probability at time $t$.  

To find a choice with this special property, we order the covariate index differences across time by choice.  
In particular,  find the choice with the largest change in covariate index:
\begin{equation} 
d^* \ = \ argmax_{c } \ \ \left( x_{c,s}^\prime\theta_0 \;  - \;  x_{c,t}^\prime \theta_0 \right). 
\label{eqn:d*}
\end{equation} 
If there is more than one choice in the $argmax$ set, then set $d^*$ to any element 
of this set.  Note that 
%\[ 
%x_{d^*,s}^\prime\theta_0 \;  - \;  x_{d^*,t}^\prime \theta_0 \ \ge \ 
%x_{c,s}^\prime\theta_0 \;  - \;  x_{c,t}^\prime \theta_0, \ \ \forall c  \ \ \ 
%\Longrightarrow \ \ \ 
%x_{c,t}^\prime \theta_0 \;  - \;  x_{d^*,t}^\prime \theta_0 \ \ge \ 
%x_{c,s}^\prime\theta_0 \;  - \;    x_{d^*,s}^\prime\theta_0,  \ \ \forall c
%\] 
\[ 
x_{d^*,s}^\prime\theta_0 \;  - \;  x_{d^*,t}^\prime \theta_0 \ \ge \ 
x_{c,s}^\prime\theta_0 \;  - \;  x_{c,t}^\prime \theta_0, \ \ \forall c  \ \ \ 
\] 
\begin{equation}
\label{eqn:indmax}
 \ \Longrightarrow \ \ \ 
x_{c,t}^\prime \theta_0 \;  - \;  x_{d^*,t}^\prime \theta_0  \;  + \;  (\lambda_c  \;  - \;  \lambda_{d^*}) \ge \ 
x_{c,s}^\prime\theta_0 \;  - \;    x_{d^*,s}^\prime\theta_0  \;  + \;  (\lambda_c  \;  - \;  \lambda_{d^*}),  \ \ \forall c
\end{equation} 
The latter covariate index differences on either side of the inequality are the same 
differences that  define $\mathcal{E}_{d^*,t}$ and $\mathcal{E}_{d^*,s}$ in (\ref{eqn:reg}).  
And the inequality (\ref{eqn:indmax}) ensures that 
\[ 
\mathcal{E}_{d^*,t} \subset \mathcal{E}_{d^*,s}
\] 
Hence, 
\begin{eqnarray} 
\Pr(y_s = d^*  \, | \, x_s, x_t,\lambda) &=& 
 \Pr( \varepsilon_s \ \in \ \mathcal{E}_{d^*,s}\, | \, x_s, x_t,\lambda ) 
 \nonumber 
\\ 
&=& 
 \Pr( \varepsilon_t \ \in \ \mathcal{E}_{d^*,s}\, | \, x_s, x_t,\lambda ) 
 \nonumber 
\\ 
&\ge& 
 \Pr( \varepsilon_t \ \in \ \mathcal{E}_{d^*,t}\, | \, x_s, x_t,\lambda )
 \nonumber
\\ 
&=& 
\Pr(y_t = d^*  \, | \, x_s, x_t,\lambda). \label{eqn:initial} 
\end{eqnarray} 
The first and last equalities follow from the definition of the disturbance regions in (\ref{eqn:dreg}).  The second equality follows from 
Assumption~\ref{asm:orthog}, and the inequality follows from the set inclusion derived above.  
Below we will extend the argument behind this inequality to generate additional conditional choice probability comparisons. 
Conditional probability inequalities like the one above can be 
translated into corresponding conditional moment inequalities, which, in turn, can be considered for identification of 
the parameter $\theta_0$.  


%Figure~\ref{fig1} illustrates the key intuition behind this inequality.  
To illustrate the key intuition behind this inequality, consider the case with three choices and $d^* =2$ implying 
that $\mathcal{E}_{2,t} \subset \mathcal{E}_{2,s}$. 
 In Figure~\ref{fig1a}, the time $s$ disturbance space for $\varepsilon_s$ 
is partitioned into regions $\mathcal{E}_{0,s}$, 
$\mathcal{E}_{1,s}$, and $\mathcal{E}_{2,s}$ corresponding to $y_s = 0, 1, 2$, respectively, 
and the vertical grey lines highlight region $\mathcal{E}_{2,s}$.  A similar 
partitioning of the disturbance space at time $t$ is shown in Figure~\ref{fig1b}, and horizontal grey lines highlight region $\mathcal{E}_{2,t}$.  Since $\varepsilon_s$ and $\varepsilon_t$ 
share the same (conditional) distribution and hence the same support set, these regions can be 
usefully superimposed in Figure~\ref{fig1c}.  
From equation (\ref{eqn:reg}), the regions $\mathcal{E}_{0,t}$, 
$\mathcal{E}_{1,t}$, and $\mathcal{E}_{2,t}$ are a translation shift of the regions $\mathcal{E}_{0,s}$, 
$\mathcal{E}_{1,s}$, and $\mathcal{E}_{2,s}$ by 
$(x_{1,s}^\prime\theta_0   -   x_{1,t}^\prime \theta_0, x_{2,s}^\prime\theta_0   -   x_{2,t}^\prime \theta_0)$. 
The size of these shifts is apparent in Figure~\ref{fig1c}, and, in the case illustrated,  
 $x_{2,s}^\prime\theta_0   -   x_{2,t}^\prime \theta_0 \ge x_{1,s}^\prime\theta_0   -   x_{1,t}^\prime \theta_0 \ge 0$, 
it follows from (\ref{eqn:d*}) that $d^* =2$.  
Starting from the nexus of the regions in Figure~\ref{fig1a}, the direction of the translation shift is interior to $\mathcal{E}_{2,s}$, which 
ensures that  $\mathcal{E}_{2,t} \subset \mathcal{E}_{2,s}$.  





\begin{comment} 
\begin{figure}
\caption{Disturbance Regions for 3 Choices}
\includegraphics[scale=.9]{sharp_id_graph1_2017.pdf}
\end{figure} 

\end{comment} 




\begin{figure}
    \centering
     \caption{Disturbance Regions for 3 Choices\\[1.2ex] $\Delta x_2^\prime \theta_0  \ge \Delta x_1^\prime \theta_0 \ge 0 \ \ \ \Longrightarrow \ \ \ \mathcal{E}_{2,t} \subset \mathcal{E}_{2,s}$}
    \begin{subfigure}[b]{0.3\textwidth}
   %     \includegraphics[width=\textwidth]{sharp_id_graph1a_2017.pdf}
    \includegraphics[scale=.55]{sharp_id_graph1a_2017.pdf}
    \vspace{-3.9in}
   %     \caption{$\mathcal{E}_{2,s} = \{\varepsilon_s: y_s=2\}$ (vertical lines)}
                  \caption{~ }
        \label{fig1a}
    \end{subfigure}
    ~ %add desired spacing between images, e. g. ~, \quad, \qquad, \hfill etc. 
      %(or a blank line to force the subfigure onto a new line)
    \begin{subfigure}[b]{0.3\textwidth}
  %      \includegraphics[width=\textwidth]{sharp_id_graph1b_2017.pdf}
   \includegraphics[scale=.55]{sharp_id_graph1b_2017.pdf}
    \vspace{-3.9in}
                \caption{~ }
        \label{fig1b}
    \end{subfigure}
    ~ %add desired spacing between images, e. g. ~, \quad, \qquad, \hfill etc. 
    %(or a blank line to force the subfigure onto a new line)
    \begin{subfigure}[b]{0.3\textwidth}
%        \includegraphics[width=\textwidth]{sharp_id_graph1c_2017.pdf}
  \includegraphics[scale=.55]{sharp_id_graph1c_2017.pdf}
   \vspace{-3.9in}
                   \caption{~ }
        \label{fig1c}
    \end{subfigure}
 %   \vspace{-2in}
   \label{fig1}
\end{figure}











%%%%%%%%%%%%%%%
%
%   GENL MOM INEQS
%
%%%%%%%%%%%%%%%

\subsection{Implied Moment Inequalities} 
\label{sec:imp}

The probability inequality in (\ref{eqn:initial}) is based on the
choice that maximizes the difference of covariate index functions.  We can push
this logic further to obtain similarly motivated inequalities based on
a complete rank ordering of the covariate index function differences over the
choices.  For time periods $s$ and $t$, start by ordering the
difference of index functions  by choice.  This ordering can be used to partition 
the choices into a set of choices with larger index function differences 
and a set with smaller index function differences.  For each such partition 
 generated by index function differences based on the true parameter value 
$\theta_0$,  we will be able to generate corresponding choice probability 
inequalities.  


Given a value of $\theta$ and vectors of covariates $x_s$ and $x_t$, we can partition the set of choices 
into two subsets corresponding to choices with larger and smaller index function differences.  
For instance, suppose $D$ is a subset of choices, i.e.\  $D\subset\{0,1,\ldots, \mathscr{D}\}$, and let 
$D^c$ denote the remaining choices, 
$D^c = \{0,1,\ldots, \mathscr{D}\}\backslash D$.  If 
 \[ 
 \min_{d\in D} \Delta x_d^\prime\, \theta 
\ge \max_{c\in D^c} \Delta x_c^\prime\, \theta,  \ \ \ \ \mbox{where}\  \Delta x_d = x_{d,s} - x_{d,t}, 
\]
then $D$ contains choices with larger index function differences and $D^c$ contains choices with 
smaller index function differences.  There are many possible partitions that could be formed in this way, and we collect the subsets corresponding to the larger index function differences as follows,
\begin{equation} 
\label{eqn:Dbar}
\overline{\mathbb{D}}(x_s, x_t, \theta) = \left\{  D\subset\{0,1,\ldots, \mathscr{D}\} \, \bigg| \,  D, D^c \ne \varnothing,  \ \min_{d\in D} \Delta x_d^\prime\, \theta 
\ge  \max_{c\in D^c} \Delta x_c^\prime\, \theta 
\right\}. 
\end{equation} 



Consider the case where no pair of choices share the same the index function difference, so that each choice has a distinct index function difference value.  
 Then $\overline{\mathbb{D}}(x_s, x_t, \theta)$ will contain a set consisting of the choice corresponding 
to the largest index function difference ($\{d^*\}$ in the notation of the previous section).  It will also contain a set 
consisting of the choices associated with the two largest index function differences, etc.  So, in this case, $\overline{\mathbb{D}}(x_s, x_t, \theta)$ 
would contain exactly $\mathscr{D}$ sets with cardinalities 1, 2, $\ldots$, $\mathscr{D}$.  Moreover for any two sets $C, D 
\in \overline{\mathbb{D}}(x_s, x_t, \theta)$ where $C$ has fewer elements than $D$, then $C \subset D$.  

In general, of course it's also possible that index function differences for some choices will be equal.  When this happens, $\overline{\mathbb{D}}(x_s, x_t, \theta)$ will contain 
more than $\mathscr{D}$ sets, and the sets in $\overline{\mathbb{D}}(x_s, x_t, \theta)$ will not be nested.  As a simple example, suppose there are four choices, $\{0, 1, 2, 3\}$.  And suppose the index function differences can be ordered as follows:  $\Delta x_3^\prime \, \theta > 
\Delta x_2^\prime \, \theta = \Delta x_1^\prime \, \theta > \Delta x_0^\prime \, \theta$ (note that $\Delta x_0=0$).  Then, $d^*=3$ and 
 $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{ \{3\}, \{3, 2\}, \{3, 1\}, \{3, 2, 1\}\}$.  
 
 \begin{comment} 
 \\[1ex] 
 {\em Example 1.} \ \   As a simple example, suppose there are $\mathscr{D}+1 = 4$ choices, $\{0, 1, 2, 3\}$.  And suppose the index function differences can be ordered as follows:  
 $$\Delta x_3^\prime \, \theta > 
\Delta x_2^\prime \, \theta = \Delta x_1^\prime \, \theta > \Delta x_0^\prime \, \theta$$ 
(Note that $\Delta x_0=0$.)  Then, $d^*=3$ and 
 $$\overline{\mathbb{D}}(x_s, x_t, \theta) = \{ \{3\}, \{3, 2\}, \{3, 1\}, \{3, 2, 1\}\}.$$ 
 \hfill $\Box$ 
 \\   
 \end{comment} 

The next result shows that the argument used to obtain a probability choice inequality for $d^*$ in the previous section can be extended to the choice sets contained in $\overline{\mathbb{D}}(x_s, x_t, \theta_0)$.  
\begin{proposition}
\label{pro:pin} 
Suppose Assumption~\ref{asm:orthog} holds.  Then, for all $D \in \overline{\mathbb{D}}(x_s, x_t, \theta_0)$, 
\[ 
\Pr( y_s \in D \, | \, x_s, x_t, \lambda) \ \ge \ \Pr( y_t \in D \, | \, x_s, x_t, \lambda).  
\] 
\vspace{-.4in} 

\hfill $\Box $
\end{proposition}
%\vskip.3cm 
%

\begin{quote} 
\noindent 
\textbf{\textsc{Proof:}} %of Proposition \ref{pro:pin}:}}   
For any set of choices $C \subset \{0, 1, \ldots, \mathscr{D}\}$ and a choice $d$, define
\begin{eqnarray*} 
\mathcal{E}_{d,C,t} &= & \left\{ \varepsilon_t:   \varepsilon_{d,t} \,\ge \, \max_{c\in C: c < d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}  
\\ && \ 
\ \cap \ \left\{ \varepsilon_t:   \varepsilon_{d,t} \, >  \, \max_{c\in C: c > d} 
\, \left[ \left( x_{c,t}^\prime\theta_0  - x_{d,t}^\prime\theta_0 \right) \, + \, \left( \lambda_c - \lambda_d \right)  + 
\varepsilon_{c,t} \right]  \right\}  
\end{eqnarray*} 
The set $\mathcal{E}_{d,C,t}$ is the region of the time $t$ disturbance space where choice $d$ is preferred 
(in random utility terms) to all choices in $C$.  Corresponding time $s$ regions $\mathcal{E}_{d,C,s}$ are defined 
analogously.  

Now take any $D \in \overline{\mathbb{D}}(x_s, x_t, \theta_0)$.  For any $d \in D$, 
$\Delta x_d^\prime \, \theta_0 \ge \Delta x_c^\prime \, \theta_0$ for all $c\in D^c$.  
Re-arranging, 
  $x_{c,t}^\prime \, \theta_0 - x_{d,t}^\prime \, \theta_0 \ \ge \  x_{c,s}^\prime \, \theta_0 - x_{d,s}^\prime \, \theta_0$ 
   for all $c\in D^c$.  It follows that $\mathcal{E}_{d,D^c,t}  \subset \mathcal{E}_{d,D^c,s}$ and this set inclusion is the main 
   step to showing the desired probability inequality:
   \begin{eqnarray}
   \Pr(y_s \in D \, | \, x_s, x_t, \lambda) &=& \Pr\left(  \varepsilon_s \in \bigcup_{d \in D} \, \mathcal{E}_{d,D^c,s} \, \bigg| \, x_s, x_t, \lambda \right) 
   \nonumber 
   \\[1.5ex]  
 &=& \Pr\left(  \varepsilon_t \in \bigcup_{d \in D} \, \mathcal{E}_{d,D^c,s} \, \bigg| \, x_s, x_t, \lambda \right) 
 \nonumber 
      \\[1.5ex]  
   &\ge &  \Pr\left(  \varepsilon_t \in \bigcup_{d \in D} \, \mathcal{E}_{d,D^c,t} \, \bigg| \, x_s, x_t, \lambda \right) 
   \label{eqn:ine} 
     \\[1.5ex]  
   &=& 
      \Pr(y_t \in D \, | \, x_s, x_t, \lambda) 
      \nonumber 
   \end{eqnarray} 
  Holding $(x_s, x_t, \lambda)$ fixed, $\{ \varepsilon_s: y_s \in D\} = \cup_{d\in D} \mathcal{E}_{d,D^c,s}$ which is the argument behind 
  the first equality.  The last equality holds similarly.  The second equality holds by Assumption~\ref{asm:orthog}.  
  Finally, the inequality in (\ref{eqn:ine}) holds since the set inclusion 
  $\mathcal{E}_{d,D^c,t}  \subset \mathcal{E}_{d,D^c,s}$ for all $d \in D$ implies 
  \begin{equation} 
  \label{eqn:sub}
  \bigcup_{d\in D} \mathcal{E}_{d,D^c,t}  \subset \bigcup_{d\in D} \mathcal{E}_{d,D^c,s}.
  \end{equation} 
%
% 
\hfill $\Box$ 
\end{quote} 

\vskip.4cm 

The probability inequalties obtained in Proposition~\ref{pro:pin} can be straightforwardly translated into 
 corresponding moment inequalities, as follows.  
 For any $D\subset\{0,1,\ldots, \mathscr{D}\}$, define $m_D(y_s, y_t) = {\bf 1}\{ y_s \in D\} - {\bf 1}\{ y_t \in D\}$.
   Then, Proposition~\ref{pro:pin} concludes 
that \\ $E[m_D(y_s, y_t) \, | \, x_s, x_t, \lambda ] \ \ge \ 0$ \ $\forall D \in \overline{\mathbb{D}}(x_s, x_t, \theta_0)$.  Taking the expectation 
with respect to $\lambda$ conditional on $x_s, x_t$ yields conditional moment inequalities expressed in terms of the observables $(y_s, y_t, x_s, x_t)$:
\begin{equation} 
\label{eqn:p1obs}
E[m_D(y_s, y_t) \, | \, x_s, x_t ] \ \ge \ 0 \ \ \ \ \ \forall D \in \overline{\mathbb{D}}(x_s, x_t, \theta_0). 
\end{equation} 
This set of conditional moment inequalities will be the key to the identification arguments that follow.  






%
%
%

%%%%%%%%%%%%%%%
%
%    SHARP ID
%
%%%%%%%%%%%%%%%

\section{Sharp Identification}
\label{sec:sharp} 

%
%
%
\begin{comment}
As noted in section~\ref{sec:pt}, the restrictions on covariates in Assumption~\ref{asm:x} are substantive.  
Part (a), for example, 
imposes lower bounds on the cardinality of discrete covariate supports and rules out the inclusion of time-varying individual 
characteristics that do not depend on choice.  Part (b) requires unboundedness 
on part of the covariate support space.    In the binary choice model, \citeasnoun{cham:2010} shows that some unboundedness 
is necessary for binary choice point identification with a general disturbance distribution.  Chamberlain's result 
can be straightforwardly extended to the multinomial choice setting to show that unboundedness 
is similarly necessary for the conclusion of Theorem~\ref{thm:pt}.  
In particular applications, any of the restrictions imposed by Assumption~\ref{asm:x} could be violated.  In this section, we 
consider the identifying power of the conditional moment inequalities  from section~\ref{sec:2} without the covariate 
restrictions of Assumption~\ref{asm:x}.

We also relax Assumption~\ref{asm:supp}.  To the extent that that the disturbances $\varepsilon_t$ represent 
unobserved time-varying determinants of choice,  Assumption~\ref{asm:supp} would not be satisfied 
if these determinants were bounded or discrete.  More generally, a key objective of the semiparametric approach 
in multinomial choice is to allow for an arbitrary joint distribution of disturbances across choices that 
could mimic the substitution patterns over choices regardless of application.  
Since Assumption~\ref{asm:supp} is the only condition 
imposed on the joint distribution of disturbances, relaxing this assumption 
achieves the semiparametric goal of complete flexibility in this aspect of the multinomial choice model.  

Relaxing Assumptions~\ref{asm:supp} and~\ref{asm:x} leaves only Assumption~\ref{asm:orthog}.  Under 
Assumption~\ref{asm:orthog}, the conditional moment inequalities that are central to the current work, 
take a weak inequality form as in Proposition~\ref{pro:pin}.  
\end{comment} 
%
%
%

The conditional moment inequalities in (\ref{eqn:p1obs}) generated by Proposition~\ref{pro:pin} 
depend 
only on the observable variables, $(y_s, y_t, x_s, x_t)$ with distribution $F_{y_s, y_t, x_s, x_t}$.  
So, the corresponding 
 identified set  can  be defined as follows:
\[ 
{\Theta}_0 =  {\Theta}_0(F_{y_s, y_t, x_s, x_t}) =\{ \theta\in\Theta \, | \, 
E[{m}_D(y_s, y_t) \, | \, x_s, x_t ] \ \ge \ 0 \ \ \forall 
D \in \overline{\mathbb{D}}(x_s, x_t, \theta) 
 \ a.s.\ (x_s, x_t)\}. 
\] 
Under Assumption~\ref{asm:orthog} alone, $\Theta_0$ will {\em not} generally be a point (up to scale).  
However, the below result shows that the conditional moment inequalities defining $\Theta_0$ do, in fact,  contain all available information about the parameter in the sense that the identified set, $\Theta_0$, provides sharp bounds on the parameter.

% define sharpness

Given random variables $(x_s, x_t, \lambda, \varepsilon_s, \varepsilon_t)$ satisfying Assumption~\ref{asm:orthog} and a value of the parameter $\theta \in \Theta$, the multinomial choice 
framework defines outcomes $y_s = \mathcal{y}(x_s, \lambda, \varepsilon_s, \theta)$ and  
$y_t = \mathcal{y}(x_t, \lambda, \varepsilon_t, \theta)$.  Then, 
the observable random variables from the multinomial choice model satisfying Assumption~\ref{asm:orthog} 
are simply $(y_s, y_t, x_s, x_t)$ with distribution $F_{y_s, y_t, x_s, x_t}$.   
We can collect all such values of the parameter and observable distributions:
\begin{eqnarray*} 
\mathcal{M} & = \ \{ (\theta, F_{y_s, y_t, x_s, x_t}) \ |  &  (x_s, x_t, \lambda, \varepsilon_s, \varepsilon_t) \ \mbox{satisfies  Assumption}~\ref{asm:orthog}, 
\theta \in \Theta, 
\\ 
&& y_s = \mathcal{y}(x_s, \lambda, \varepsilon_s, \theta),   
y_t = \mathcal{y}(x_t, \lambda, \varepsilon_t, \theta), (y_s, y_t, x_s, x_t) \sim F_{y_s, y_t, x_s, x_t} \} 
\end{eqnarray*} 
Let $F_{y_s, y_t, x_s, x_t}$ be any observable distribution from the multinomial choice framework under 
Assumption~\ref{asm:orthog}, i.e. for some $\theta \in \Theta$, $(\theta, F_{y_s, y_t, x_s, x_t}) \in \mathcal{M}$.  
Then the 
sharp identified set is simply the projection of $\mathcal{M}$ onto $\Theta$ for the given observable distribution,
\[ 
\Theta_S = \Theta_S(F_{y_s, y_t, x_s, x_t}) = \{\theta\in\Theta   \, | \,  (\theta, F_{y_s, y_t, x_s, x_t}) \in \mathcal{M} \} 
\]

Sharpness of the identified set $\Theta_0$ is simply that $\Theta_0 = \Theta_S$.  More formally, let 
\[ 
\mathcal{F}_{ob}  = \{ F_{y_s, y_t, x_s, x_t} \, | \, (\theta, F_{y_s, y_t, x_s, x_t}) \in \mathcal{M} \ \mbox{for some}\ \theta \in\Theta\}.
\]  
So, $\mathcal{F}_{ob}$ is the set of all possible multinomial choice observable distributions generated by some distribution of unobservables 
satisfying Assumption~\ref{asm:orthog} and some parameter value $\theta \in \Theta$.  
Then, we have the following result.

\begin{theorem} 
\label{thm:sharp}
Under Assumption~\ref{asm:orthog}, $\Theta_0$ is sharp.  That is, 
\[ 
{\Theta}_0(F_{y_s, y_t, x_s, x_t}) = {\Theta}_S(F_{y_s, y_t, x_s, x_t})
\] 
for all $F_{y_s, y_t, x_s, x_t} \in \mathcal{F}_{ob}$.

\hfill $\Box$ 
\end{theorem} 


\begin{comment} 
 Let $F_{y_s, y_t, x_s, x_t} \in \mathcal{F}_{ob}$.  
First, note that $\Theta_S \subset \Theta_0$.  For 
 $\theta \in 
{\Theta}_S(F_{y_s, y_t, x_s, x_t})$, $(\theta, F_{y_s, y_t, x_s, x_t}) \in \mathcal{M}$.  By Proposition~\ref{pro:pin}, 
$E[{m}_D(y_s, y_t, x_s, x_t, \theta) \, | \, x_s, x_t ] \ \ge \ 0$, \ $\forall D \in \overline{\mathbb{D}}(x_s, x_t, \theta)$.  Hence 
$\theta \in {\Theta}_0(F_{y_s, y_t, x_s, x_t})$.  
\end{comment} 

Theorem~\ref{thm:sharp} is shown by a constructive proof, see Appendix for details.    
Fix any distribution of observables, $F_{y_s, y_t, x_s, x_t} \in \mathcal{F}_{ob}$.  Let $\Theta_0$ and $\Theta_S$ denote the 
identified sets associated with this observable distribution, ${\Theta}_0={\Theta}_0(F_{y_s, y_t, x_s, x_t})$ and ${\Theta}_S= {\Theta}_S(F_{y_s, y_t, x_s, x_t})$.  
  It is straightforward to establish that $\Theta_S \subset \Theta_0$ 
(using Proposition~\ref{pro:pin}).  So, 
the main argument for Theorem~\ref{thm:sharp} is to show the set inclusion in the other direction, $\Theta_0 \subset \Theta_S$.  

\begin{comment}
{\color{blue} Jack we probably disagree here, but I would note the following.  If $F_{y_s, y_t, x_s, x_t}=F^e_{y_t, y_s, x_s, x_t}$ where $F^e_{y_s, y_t, x_s, x_t}$ denotes
the distribution of choices given the observed combinations of $(x_s,x_t)$ in a particular data set, then the notion of sharpness considered here corresponds to a different limiting
argument often implicit in applied discussions of the limited power of a particular data set.  It would correspond to the limiting distribution of a data set which was generated
by drawing from the $(x_s,x_t)$ distribution found in the particular data set being considered.}
\end{comment} 

Take any $\theta \in \Theta_0$.  We exhibit a conditional distribution $(\lambda^*, \varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t $ such that 
$(x_s, x_t, \lambda^*, \varepsilon^*_s, \varepsilon^*_t)$ satisfies Assumption~\ref{asm:orthog} and $(y^*_s, y^*_t, x_s, x_t) \sim F_{y_s, y_t, x_s, x_t}$
where $y^*_s = \mathcal{y}(x_s, \lambda^*, \varepsilon^*_s, \theta)$ and $y^*_t = \mathcal{y}(x_t, \lambda^*, \varepsilon^*_t, \theta)$.  
Then, $(\theta, F_{y_s, y_t, x_s, x_t}) = (\theta, F_{y^*_s, y^*_t, x_s, x_t}) \in \mathcal{M}$ and so $\theta\in\Theta_S$.  


Let $(x_s, x_t )$ be any pair of covariate values in the support of the joint distribution $\mathcal{X}_{s,t}$.  
We need to choose $(\lambda^*, \varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t $ such that $\Pr(y^*_s = d, y^*_t=d' \, | \, x_s, x_t) = 
\Pr(y_s = d, y_t=d' \, | \, x_s, x_t)$.  Setting $\lambda^*=0$, $\Pr(y^*_s = d, y^*_t=d' \, | \, x_s, x_t)$ is determined by the behavior of 
$(\varepsilon^*_s, \varepsilon^*_t) \, | \,  x_s, x_t$ on certain ``choice-determining'' regions of  $\mathbb{R}^{2\mathscr{D}}$.  
We further subdivide these regions so that symmetry can be imposed on the corresponding discrete marginal distributions to satisfy  Assumption~\ref{asm:orthog}.  



Let $R_{d,d'} = \{\varepsilon \, | \, \mathcal{y}(x_s, \theta, \lambda^*=0, \varepsilon) = d\} \, \cap \, 
\{\varepsilon \, | \,\mathcal{y}(x_t, \theta, \lambda^*=0, \varepsilon) = d'\}$.  
%$R_{d,d'} = \{\varepsilon \, | \, \mathcal{y}(x_s, \theta, \lambda^*=0, \varepsilon) = d, \ \mathcal{y}(x_t, \theta, \lambda^*=0, \varepsilon) = d'\}$.  
These sets can be used   to subdivide the choice-determining sets on $\varepsilon^*_s \, | \,  x_s, x_t$ and to similarly subdivide the choice-determining sets on $\varepsilon^*_t \, | \,  x_s, x_t$.  
That is, 
\[ 
\{\varepsilon^*_s \, | \, \mathcal{y}(x_s, \theta, \lambda^*=0, \varepsilon^*_s) = d\} \, = \, \bigcup_{d'=0}^{\mathscr{D}}   R_{d,d'} \ \ 
\mbox{and} \ \ \{\varepsilon^*_t \, | \, \mathcal{y}(x_t, \theta, \lambda^*=0, \varepsilon^*_t) = d'\} \, = \, \bigcup_{d=0}^{\mathscr{D}}   R_{d,d'}.
\] 
So $\Pr(y^*_s = d \, | \, x_s, x_t)% = \Pr( \cup_{d'} \epsilon_s \in R_{d,d'} \, | \, x_s, x_t) 
= \sum_{d'=0}^{\mathscr{D}}  \Pr(  \varepsilon^*_s \in R_{d,d'} \, | \, x_s, x_t) $  and similarly $\Pr( y^*_t=d' \, | \, x_s, x_t)  
 % % =\Pr( \cup_{d} \epsilon_t \in R_{d,d'} \, | \, x_s, x_t) 
= \sum_{d=0}^{\mathscr{D}}  \Pr(  \varepsilon^*_t \in R_{d,d'} \, | \, x_s, x_t)$.  
From these expressions, we see that the sets $R_{d,d'}$ can be used to describe the marginal behavior of $y^*_s$ (and $y^*_t$), and in fact provide a device for imposing the homogeneity in $\varepsilon^*_t$ across time $t$ as required by Assumption~\ref{asm:orthog}(b).  

Additionally, the Cartesian products of sets of the form $R_{d,d'}$ can be used to describe the joint 
behavior of $y^*_s$ and $y^*_t$. 
So, 
%\[ 
%\{(\varepsilon^*_s, \varepsilon^*_t) \, | \, \mathcal{y}(x_s, \theta, \lambda^*=0, \varepsilon^*_s) = d, 
%\mathcal{y}(x_t, \theta, \lambda^*=0, \varepsilon^*_t) = d'\} \, = \, \left(  \bigcup_{d''=0}^{\mathscr{D}}   R_{d,d''} \right) 
%\times \left( \bigcup_{d'''=0}^{\mathscr{D}}   R_{d''',d'}\right) 
%\] 
%and 
\[ 
\Pr(y^*_s = d, y^*_t = d' \, | \, x_s, x_t) = \sum_{d''=0}^{\mathscr{D}} \sum_{d'''=0}^{\mathscr{D}} 
\Pr(  (\varepsilon^*_s, \varepsilon^*_t) \in R_{d,d''} \times R_{d''',d'}\, | \, x_s, x_t). 
\] 
Now,  let 
  $q^*_{d,d' \times d'', d'''} = \Pr( (\varepsilon^*_s, \varepsilon^*_t) \in 
R_{d,d'} \times R_{d'',d'''}  \, | \,  x_s, x_t)$.  
Then, we can translate our problem of exhibiting a conditional distribution $(\lambda^*, \varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t $ that both satisfies Assumption~\ref{asm:orthog} and generates 
an observable distribution matching $F_{y_s, y_t, x_s, x_t}$ into a problem of finding a solution 
to the  
 system of linear equations below.  Here, we treat $\Pr(y_s = d, y_t=d' \, | \, x_s, x_t)$ 
 as known and seek a nonnegative solution for $q^*_{d,d'' \times d', d'''}$:  
\begin{align}
\label{eqn:q1}
  & \Pr(y_s = d, y_t=d' \, | \, x_s, x_t)  \, = \,  \sum_{d''=0}^{\mathscr{D}} \sum_{d'''=0}^{\mathscr{D}}  q^*_{d,d'' \times d''', d'}  \\ 
%\mbox{(**)} 
\label{eqn:q2} 
&  \sum_{d''=0}^{\mathscr{D}} \sum_{d'''=0}^{\mathscr{D}}  q^*_{d,d' \times d'', d'''} \, = \,   \sum_{d''=0}^{\mathscr{D}} \sum_{d'''=0}^{\mathscr{D}}  q^*_{d'',d''' \times d, d'}
\end{align} 
If $q^*_{d,d' \times d'', d'''}$ satisfies (\ref{eqn:q1}), then $(y^*_s, y^*_t, x_s, x_t) \sim F_{y_s, y_t, x_s, x_t}$, as desired.  If $q^*_{d,d' \times d'', d'''}$ satisfies (\ref{eqn:q2}), then the conditional disturbance distribution can be constructed to satisfy Assumption~\ref{asm:orthog}.  A simple way to 
achieve this construction is to choose a point in each region, $r_{d,d'} \in R_{d,d'}$.  Then choose the distribution of 
$(\varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t $ to be discrete with  $\Pr( (\varepsilon^*_s, \varepsilon^*_t) = 
(r_{d,d'}, r_{d'',d'''})  \, | \,  x_s, x_t) = q^*_{d,d' \times d'', d'''}$.  

The last step is to show existence of  a nonnegative solution for $q^*_{d,d' \times d'', d'''}$ satisfying (\ref{eqn:q1}) and (\ref{eqn:q2}).  
Existence is established for an equivalent dual problem, using Farkas' Lemma, see Appendix.


There are several interesting features of the constructed distribution that proves Theorem~\ref{thm:sharp}.  First, there are additional constraints not 
apparent in equations (\ref{eqn:q1}) and (\ref{eqn:q2}).  In particular, from the proof of Proposition~\ref{pro:pin}, we know that some of 
the regions $R_{d,d'}$ will be empty, and so any corresponding $q^*_{d,d' \times d'', d'''}$ or $q^*_{d'',d''' \times d, d'}$ will be zero.  
For example, suppose $d^*$ is the covariate index difference maximizing choice based on $\theta$, 
as defined in section~\ref{sec:1mom}.  Then, 
$\{\varepsilon^*_t \, | \, \mathcal{y}(x_t, \theta, \lambda^*=0, \varepsilon^*_t) = d^*\} \, 
\subset \, \{\varepsilon^*_s \, | \, \mathcal{y}(x_s, \theta, \lambda^*=0, \varepsilon^*_s) = d^*\}$. 
The choice-determining set for $d^*$ at time $t$ is contained in the choice-determining 
set for $d^*$ at time $s$, which is the basic idea behind the first conditional probability inequality derived 
in (\ref{eqn:initial}).  
Fixing $d^*$ to be defined as in section~\ref{sec:1mom}, it  follows that for $d \ne d^*$, $R_{d,d^*} = \varnothing$.  
Additional constraints of this type have to be accounted for in the solutions to (\ref{eqn:q1}) and (\ref{eqn:q2}).  

Second, the fixed effects are set to zero in the constructed distribution and essentially play no role.  
Fixed effect variation is not needed to match the simulated distribution to the given distribution of observables.  
This feature also reflects the fact that only within variation is used in the identification through the conditional moment inequalities.  

Third, 
% it is not apparent from the overview of the construction argument above, but 
note that some across time correlation in the 
distribution of $(\varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t$ is, in general, needed to solve equations (\ref{eqn:q1}) and (\ref{eqn:q2}).  
Notice that satisfaction of Assumption~\ref{asm:orthog} is achieved by matching the distributions of 
$\varepsilon^*_s \, | \, x_s, x_t $ and $\varepsilon^*_t \, | \, x_s, x_t$.  However, the  full flexibility from the conditional joint distribution 
$(\varepsilon^*_s, \varepsilon^*_t) \, | \, x_s, x_t$ is needed to match the simulated distribution to the given observable distribution, which can include correlation between choices across time.  
In particular, if we impose conditional independence between $\varepsilon^*_s$ and $\varepsilon^*_t$ in the simulated distribution, 
then a solution to (\ref{eqn:q1}) and (\ref{eqn:q2}) will not always exist. 


Fourth,  the proof outlined above constructs a distribution of unobservables conditional on $(x_s, x_t)$.  
The same constructed distribution could be used to exhibit the equivalence of $\Theta_0$ and $\Theta_S$ for other 
``designs,'' i.e.\ covariate distributions.  For example, given observed combinations of $(x_s, x_t)$  
in a particular data set, let 
 $F^e_{y_s, y_t, x_s, x_t}$ denote the corresponding distribution of choices and covariates for this empirical distribution of covariates.   
 Then,  $F^e_{y_s, y_t, x_s, x_t}$ can be used to define an identified set $\Theta_0^e$ and a sharp identified set $\Theta^e_S$.  By changing only the marginal distribution of covariates and maintaining the same conditional distribution of unobservables to generate $F^e_{y_s, y_t, x_s, x_t}$, Assumption~\ref{asm:orthog} is still satisfied.    The conclusion of Theorem~\ref{thm:sharp} follows.  Practically, this means that to consider the identifying ``power'' of alternative covariate designs, the conditional moment inequalities defining $\Theta_0$  
 still suffice to provide sharp conclusions.  
 




%
%
%



%%%%%%%%%%%%%%%
%
%    ADDITIONAL REMARKS
%
%%%%%%%%%%%%%%%



\section{Additional Remarks}
\label{sec:add}

Next, we discuss topics related to our sharp identification result, including point identification, the special case of binary choice, and estimation and inference.

\subsection{Point Identification}
\label{sec:pt}

In section~\ref{sec:sharp}, we showed that the proposed conditional moment inequalities produce a sharp identified set.  Point identification is established by imposing further conditions that ensure the identified set 
reduces to a singleton.  Assumption~\ref{asm:orthog} places no restrictions on  the covariates.  The key additional conditions for point identification ensure sufficient variation in the covariates, in particular an assumption of unboundedness (see \cite{cham:2010}).

\citeasnoun{shi:shu:son:2018} derived conditional moment inequalities implied by cyclic monotonicity for the multinomial choice model under conditions including Assumption~\ref{asm:orthog} and absolute continuity of the error distribution with respect to Lebesgue measure.   Under assumptions on the covariates, they show that their conditional moment inequalities are sufficient for point identification.  

It is straightforward to compare the conditional moment inequalities (\ref{eqn:p1obs}) derived above to the corresponding \citeasnoun{shi:shu:son:2018} cyclic monotonicity conditional moment inequalities.  From 
\citeasnoun{shi:shu:son:2018} Lemma 3.1, the length 2-cycle conditional moment inequality can be expressed as 

\begin{equation}
\label{eqn:sss1} 
0 \ \le \ \sum_{d=1}^{\mathscr{D}} \left[ \Pr(y_{i,s} =d \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =d \, | \, x_{i,s}, x_{i,t}) 
\right] \Delta x_d^\prime \theta_0
\end{equation} 
Now denote a (weak) ordering of covariate index differences as follows:
\begin{equation} 
\label{eqn:wor}
\Delta x_{(\mathscr{D})^*}^\prime \theta_0 \ \ge \ \Delta x_{(\mathscr{D}-1)^*}^\prime \theta_0  
 \ \ge \ \cdots  \ \ge \  \Delta x_{(0)^*}^\prime \theta_0
\end{equation} 
And suppose that choice zero has the $j+1^{th}$ smallest covariate index difference, i.e. $(j)^* \ = \ 0$, so that  $\Delta x_{(j)^*}^\prime \theta_0 \ = \ 0$.  Then, re-writing the sum in (\ref{eqn:sss1}),
%\begin{adjustwidth}{-2cm}{-1cm}
\begin{eqnarray} 
\lefteqn{ 
\sum_{d=1}^{\mathscr{D}} \Big[ \Pr(y_{i,s} =d \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =d \, | \, x_{i,s}, x_{i,t}) 
\Big]  \Delta x_d^\prime \theta_0 
} 
\nonumber 
\\ 
& \hspace*{-21mm} =&   \hspace*{-12mm} 
\sum_{d=j+1}^{\mathscr{D}}\Big[ \Pr(y_{i,s} =(d)^* \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =(d)^* \, | \, x_{i,s}, x_{i,t}) 
\Big]  \Delta x_{(d)^*}^\prime \theta_0  
\nonumber
\\ 
&&  \hspace*{-15mm} 
+ 
\sum_{d=0}^{j-1} \Big[ \Pr(y_{i,s} =(d)^* \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =(d)^* \, | \, x_{i,s}, x_{i,t}) 
\Big]  \Delta x_{(d)^*}^\prime \theta_0  
\nonumber
\\ 
& \hspace*{-21mm} =&   \hspace*{-12mm} 
\sum_{d=j+1}^{\mathscr{D}} \left[ \sum_{d^\prime=d}^{\mathscr{D}} \left( 
\Pr(y_{i,s} =(d^\prime)^* \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =(d^\prime)^* \, | \, x_{i,s}, x_{i,t})  \right)
\right] 
\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d-1)^*}^\prime \theta_0  \right) 
\nonumber
\\ 
&&  \hspace*{-15mm} 
+ 
\sum_{d=0}^{j-1} \left[ \sum_{d^\prime=d}^{\mathscr{D}} \left( 
\Pr(y_{i,s} =(d^\prime)^* \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} =(d^\prime)^* \, | \, x_{i,s}, x_{i,t})  \right)
\right] 
\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d+1)^*}^\prime \theta_0  \right) 
\nonumber
\\ 
& \hspace*{-21mm} =&   \hspace*{-12mm}  
\sum_{d=j+1}^{\mathscr{D}} \Big[ 
\Pr(y_{i,s}  \in \{ (d)^*, \ldots, (\mathscr{D})^*\} \, | \, x_{i,s}, x_{i,t}) 
-  \Pr(y_{i,t}  \in \{ (d)^*, \ldots, (\mathscr{D})^*\} \, | \, x_{i,s}, x_{i,t}) 
\Big] 
\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d-1)^*}^\prime \theta_0  \right) 
\nonumber
\\ 
&&  \hspace*{-15mm} 
+ 
\sum_{d=0}^{j-1} \Big[
\Pr(y_{i,s}  \in \{(0)^*, \ldots,  (d)^*\} \, | \, x_{i,s}, x_{i,t}) - \Pr(y_{i,t} \in \{(0)^*, \ldots,  (d)^*\}  \, | \, x_{i,s}, x_{i,t})  
\Big]  
\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d+1)^*}^\prime \theta_0  \right) 
\nonumber
\\ 
& \hspace*{-21mm} =&   \hspace*{-12mm} 
\sum_{d=j+1}^{\mathscr{D}} \Big[
\Pr(y_{i,s}  \in \{ (d)^*, \ldots, (\mathscr{D})^*\} \, | \, x_{i,s}, x_{i,t}) 
-  \Pr(y_{i,t}  \in \{ (d)^*, \ldots, (\mathscr{D})^*\} \, | \, x_{i,s}, x_{i,t}) 
\Big] 
\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d-1)^*}^\prime \theta_0  \right) 
\nonumber
\\ 
&& \hspace*{-15mm} 
+ 
\sum_{d=0}^{j-1} \Big[
\Pr(y_{i,s}  \in  \{ (d+1)^*, ..., (\mathscr{D})^*\}  |  x_{i,s}, x_{i,t}) - \Pr(y_{i,t} \in  \{ (d+1)^*, ..., (\mathscr{D})^*\} | x_{i,s}, x_{i,t})  
\Big] 
\left( 
\Delta x_{(d+1)^*}^\prime \theta_0  - \Delta x_{(d)^*}^\prime \theta_0  \right) 
\nonumber 
\\ 
&& ~
\label{eqn:sss2} 
\end{eqnarray} 
%
%\end{adjustwidth}
%
The terms $\left( 
\Delta x_{(d)^*}^\prime \theta_0  - \Delta x_{(d-1)^*}^\prime \theta_0  \right)$ and 
$\left( 
\Delta x_{(d+1)^*}^\prime \theta_0  - \Delta x_{(d)^*}^\prime \theta_0  \right) $ in (\ref{eqn:sss2}) are nonnegative 
by the ordering defined in (\ref{eqn:wor}). The relationship between the conditional moment inequalities in 
(\ref{eqn:p1obs}) and the cyclic monotonicity conditional moment inequalities in (\ref{eqn:sss1}) follows immediately.  The inequalities in (\ref{eqn:p1obs}) imply that the probability difference terms in square brackets in 
(\ref{eqn:sss2}) are nonnegative, which further implies that  (\ref{eqn:sss1})  holds.  It is also clear from the 
expression in (\ref{eqn:sss2}) that the implication does not go in the other direction.  That is, inequalities 
 (\ref{eqn:sss1}) do {\em not} imply  (\ref{eqn:p1obs}). 

We conclude the under Assumption~\ref{asm:orthog}, the conditional moment inequalities in (\ref{eqn:p1obs}) yield sharp identification and under the additional conditions given in \citeasnoun{shi:shu:son:2018}, 
these inequalities also yield point identification.\footnote{Actually, we find that point identification can be 
achieved using a subset of the inequalities in 
(\ref{eqn:p1obs}) that correspond to partitions of the choice set of a fixed size.  Fix $\delta \in \{1, \ldots, 
\mathscr{D}\}$.  The, for point identification, it suffices to consider the subset of condition moment inequalities in 
(\ref{eqn:p1obs}) with $|D| = \delta or (\mathscr{D}+1)-\delta$.  This collection of condition moment inequalities is non-nested with the cyclic monotonicity inequalities in (\ref{eqn:sss1}).  
}












%%%%%%%%%%%%%%%
%
%    MAX SCORE
%
%%%%%%%%%%%%%%%

\subsection{Binary Choice}
\label{sec:ms} 

The sharpness result in  section~\ref{sec:sharp} can, of course, be specialized to the case of binary choice.  
In this case, Theorem~\ref{thm:sharp} provides a (to our knowledge) new supplement to the point identification finding in 
\citeasnoun{mans:1987}.  In particular, even when point identification fails, 
the weak version of \citeasnoun{mans:1987}'s conditional moment inequalities provide sharp bounds on the 
binary choice random utility covariate coefficient, $\theta$, under Assumption~\ref{asm:orthog}.\footnote{Assumption~\ref{asm:orthog}(a) 
is exactly Manski's Assumption 3.  Assumption~\ref{asm:orthog}(b) is exactly Manski's Assumption 1(a).  So, the sharpness result 
relaxes Manski's Assumptions 1(b) and 2.
% (which correspond to Assumptions~\ref{asm:supp} and~\ref{asm:x} in the current work).
}  
Moreover, we show here that Manski's maximum score criterion used for point identification can also 
be used to obtain sharp partial identification when point identification does not hold.  






In the point identified binary choice case, \citeasnoun{mans:1987} proposes an alternative method of estimation, commonly 
referred to as maximum score estimation.  We  see below the close connection between the maximum score criterion and the conditional moment inequalities 
defining the identified set.  

The conditional moment inequalities are:
 \begin{equation}
 \label{eqn:md}
E[{m}_D(y_s, y_t) \, | \, x_s, x_t ] \ \ge \ 0, \ \ \ \ \forall\ D \in \overline{\mathbb{D}}(x_s, x_t, \theta)
\end{equation}
where in the binary choice case, as 
noted in section~\ref{sec:imp}, 
 \begin{equation}
 \label{eqn:mdD}
 \overline{\mathbb{D}}(x_s, x_t, \theta) = \left\{ \begin{array}{cl} \{\{1\}\} & \ \mbox{for}\ \Delta x_1^\prime \theta > 0 
 \\ 
 \{\{0\}\} & \ \mbox{for}\ \Delta x_1^\prime \theta < 0 
  \\ 
 \{\{0\}, \{1\}\} & \ \mbox{for}\ \Delta x_1^\prime \theta = 0. 
 \end{array} 
 \right. 
\end{equation}
And, the maximum score criterion function is:
\[ 
H(\theta) = E[ \mbox{sgn}( \Delta x_1^\prime \theta) (y_s - y_t)].
\] 
To connect these expressions, define a function which ``aggregates'' the conditional moments across the sets $D\in \overline{\mathbb{D}}(x_s, x_t, \theta)$:
 \begin{equation}
 \label{eqn:Hx}
 H(x_s, x_t, \theta) = \sum_{ D \in \overline{\mathbb{D}}(x_s, x_t, \theta)}
 E[{m}_D(y_s, y_t) \, | \, x_s, x_t ].  
\end{equation}
Then,
 \begin{equation}
 \label{eqn:H}
H(\theta) = E[H(x_s, x_t, \theta)].
\end{equation}
That is, the maximum score criterion is an aggregation of the {\em un}conditional version of the  moments from the conditional moment inequalities.  

Clearly, $H(\theta_0) \ge 0$ and $\Theta_0 \subset \{\theta \in \Theta \, | \, H(\theta)\ge 0\}$, but in general this set inclusion would be strict and 
$\{\theta \in \Theta \, | \, H(\theta)\ge 0\}$ would not be sharp.  Instead of checking non-negativity of this criterion,  maximum score seeks to maximize it.  
According to 
\citeasnoun{mans:1987} in the binary choice case, under conditions implying point identification, the maximum score criterion $H(\theta)$ is uniquely maximized at $\theta_0$.   

The following proposition shows that, in the binary choice case, the maximum score criterion is useful even when point identification is 
not achieved.  In particular,  under Assumption~\ref{asm:orthog} alone, $\Theta_0$ could be either a set or a point, and %we find that 
the maximum score criterion $H(\theta)$ exactly identifies this set (or point) $\Theta_0$.  

\begin{proposition}
\label{pro:ms}
Suppose $\mathscr{D}+1 =2$ (binary choice) and  Assumption~\ref{asm:orthog} holds.  Then,
\[ 
\Theta_0 \ = \ \arg\max_{\theta\in\Theta} H(\theta) \ = \ \{\theta \in \Theta: H(\theta) = H(\theta_0)\}.
\] 
\hfill $\Box$
\end{proposition} 
%
% 
Under the conditions of Theorem~\ref{thm:sharp}, $\Theta_0$ is itself sharp, 
 showing that the maximum score criterion will identify the sharp bounds on the parameter in the binary choice model. 




\subsection{Estimation and Inference}

Though our focus is on identification in the panel data multinomial choice model, 
it is worth noting that estimation and inference 
 for the identified set, $\Theta_0$, can proceed via conditional moment inequality methods 
recently developed in \citeasnoun{andr:shi:2013}, \citeasnoun{arms:2015}, \citeasnoun{arms:chan:2016}, \citeasnoun{cher:lee:rose:2013}, \citeasnoun{chet:2018}, and \citeasnoun{lee:song:whan:2013}.  Additionally, 
\citeasnoun{kha:pon:tam:2020} proposes a two-step inference approach designed for moments that are functions of probabilities.   These methods provide valid inference without requiring 
prior knowledge of whether the identified set is a point.  




%\citeasnoun{kha:pon:tam:2020} suggests a method of partial identification inference that is particularly well %suited to the multinomial choice setting with discrete covariates.  Rather than treating the moments as %generic expectations, \citeasnoun{kha:pon:tam:2020} proposes a two-step approach designed for moments %that are functions of probabilities.  In multinomial choice settings, one can often be dealing with certain %choice probabilities that are near zero, and  methods that impose the nonnegativity property of choice %probabilities can be particularly helpful in these cases.






%%%%%%%%%%%%%%%
%
%    CONCLUSION
%
%%%%%%%%%%%%%%%

\section{Conclusion}
\label{sec:con}

We have provided a new approach to identification for multinomial
choice models.  Our focus has been on the multinomial choice model which allows for
choice-specific fixed effects with a group (or panel) structure and a
nonparametric distribution of disturbances only restricted to satisfy
a stationarity assumption.   We show that this specification generates
conditional moment inequalities which can be used for identification of the parameter vector 
and avoids the incidental parameter problem using only two time periods.  
These conditional moment inequalities 
 provide sharp bounds 
without restrictions on the covariates.  



\begin{comment} 
The derived conditional moment inequalities do not depend on the linearity of the covariate 
index function.  So, even when the covariate index function takes a known parametric nonlinear form, 
 these same inequalities could used to generate an identified set.  The point and sharp identification 
arguments would, in general, not carryover to this nonlinear case, but valid inference on the partially identified 
 parameters of the nonlinear index function would be straightforward.  

 
 

One conclusion from our identification findings is that robust methods of estimation and inference 
using our conditional moment inequalities have the virtue of yielding sharp bounds  whether or not point identification holds.
 However, if point identification is known to hold, then one could pursue specialized estimation and inference methods 
 for this case.  From section~\ref{sec:ms}, the maximum score criterion will not generally work for this purposes, but 
 a criterion that penalizes violations of the conditional moment inequalities is possible.  
 \citeasnoun{arad:rose:2016} and \citeasnoun{kahn:ouya:tame:2019} have proposed estimation 
 methods for point identified conditional moment inequality models, but these papers focus on cases where the 
 contact set has positive measure and parametric rate consistency is possible.  The multinomial choice model will generally 
 only have a measure zero contact set, so these methods would need to be extended to be applied in the current setting.  
  \end{comment} 

 
 
 
Often the focus of empirical studies is not on $\theta_0$ per se but
rather on different functionals that could depend on $\theta_0$ (e.g. \cite{teba:torg:yang:2018}).  In the discrete choice panel data setting, 
\citeasnoun{cher:fern:hahn:newe:2013} suggest particular functionals of interest such as the conditional quantile or
average structural effects.  
In such cases, our conditional moment inequalities provide a new source of identifying information.  Without restricting the 
disturbance distribution across choices, our conditional moment inequalities are relatively easy to compute and 
provide sharp (and sometimes point) identifying 
information on $\theta_0$.  This additional ``within'' information can be used to improve upon  the estimation 
of the various  effects by narrowing the range of parameter values to be considered together with the possible 
 disturbance distributions (including fixed effects) that are consistent with the ``between'' variation specified in the \citeasnoun{cher:fern:hahn:newe:2013} paper.   

% In the analysis of demand systems one will often be interested in own and cross price
%elasticities, or the consumer surplus generated by changes in product
%characteristics (\cite{berr:levi:pake:2004}).
% We leave the question of
%the extent to which our results can be incorporated into the analysis
%of these and other issues for future research. 











\bibliography{multi_refs}




\clearpage 

\section{Appendix} 




\noindent 
\textbf{\textsc{Proof of Theorem~\ref{thm:sharp}:}} 
Take $F_{y_s, y_t, x_s, x_t} \in \mathcal{F}_{ob}$.  It is straightforward to show that ${\Theta}_S \subset {\Theta}_0$.  So we will focus on showing 
${\Theta}_0 \subset {\Theta}_S$.  Take $\theta \in {\Theta}_0$.  For all $x_s, x_t$ in the support of the joint covariate space, we will exhibit a conditional 
distribution $(\varepsilon^*_s, \varepsilon^*_t) | x_s, x_t$ satisfying Assumption~\ref{asm:orthog}(b) with $\lambda^*=0$ and 
$F_{y^*_s, y^*_t \, | \, x_s, x_t} = F_{y_s, y_t \, | \,  x_s, x_t}$ where $y^*_j = \mathcal{y}(x_j, \lambda^*=0, \varepsilon^*_j, \theta)$ for $j=s,t$.  

Suppose we order the covariate indices for the parameter $\theta$ and there is a strict ordering:
$\Delta x_{(\mathscr{D})}^\prime \theta \, > \, \Delta x_{(\mathscr{D}-1)}^\prime \theta \, > \, \cdots  \, > \, \Delta x_{(0)}^\prime \theta$ (using order statistic subscript notation).  Since $\theta \in 
{\Theta}_0$, the conditional moment inequalities imply \\[-1.5ex] 
\[ 
\Pr(y_s \in \{ (\mathscr{D}), \ldots, (d)\} \, | \, x_s, x_t) \ge \Pr(y_t \in \{ (\mathscr{D}), \ldots, (d)\} \, | \, x_s, x_t)  \ \ \ \forall d = 1, 2, \ldots, \mathscr{D}.
\] 
Let $p_{d,d'} = \Pr(y_s = d, y_t = d' | x_s, x_t)$ and $p^*_{d,d'} = \Pr(y^*_s = d, y^*_t = d' | x_s, x_t)$.  We need to find $(\varepsilon^*_s, \varepsilon^*_t) | x_s, x_t$ such that $p^*_{d,d'} = p_{d,d'}$ $\forall d, d'$ and $\varepsilon^*_s | x_s, x_t \  \sim \ \varepsilon^*_t | x_s, x_t$.  

Define 
$R_{d;s} = \{ \varepsilon:  \mathcal{y}(x_s, \lambda^*=0, \varepsilon^*_s, \theta) = d\}$.  The set inclusion obtained in the proof of Proposition~\ref{pro:pin} 
shows that 
\begin{equation}
\label{eqn:Rin} 
R_{(\mathscr{D});t} \cup \cdots \cup R_{(d);t} \ \subset \ R_{(\mathscr{D});s} \cup \cdots \cup R_{(d);s} \ ,\ \ \forall d \in \{1, \ldots, \mathscr{D}\}. 
\end{equation} 
Since the sets $R_{(d);s}$ form a partition for $d = 0, \ldots, \mathscr{D}$, the set inclusion (\ref{eqn:Rin}) implies that  
\begin{equation}
\label{eqn:Rem}  
R_{(d);s} \,  \cap \, R_{(d');t} = \emptyset \ \ \ \mbox{for} \ d' > d.
\end{equation}  
 Let $R_{d,d'} = R_{d;s} \cap R_{d';t}$, which is a set in the $\varepsilon^*_s$-space (or the $\varepsilon^*_t$-space).  Cartesian 
products of these sets will form sets in the $(\varepsilon^*_s,\varepsilon^*_t)$-space: let $R_{d,d' \times d'', d'''} = R_{d,d'} \times R_{d'',d'''}$ $= 
\{ (\varepsilon^*_s,\varepsilon^*_t): \varepsilon^*_s \in R_{d,d'}, \varepsilon^*_t \in R_{d'',d'''}\}$.  Finally, let 
$q^*_{d,d' \times d'', d'''} = \Pr( (\varepsilon^*_s,\varepsilon^*_t) \in R_{d,d' \times d'', d'''} | x_s, x_t )$.  These probabilities form the basic building blocks for our 
constructed $(\varepsilon^*_s, \varepsilon^*_t) | x_s, x_t$ distribution, as $ R_{d,d' \times d'', d'''}$ partitions the $(\varepsilon^*_s,\varepsilon^*_t)$-space. 
% To show sharpness, we need show that there exists $q^*_{d,d' \times d'', d'''} $ so that 
Given (\ref{eqn:Rem}), 
\[ 
p^*_{(d),(d')} \ = \ \sum_{\underaccent{\tilde}{d} = 0}^{\mathscr{D}} \sum_{\tilde{d} = 0}^{\mathscr{D}}  q^*_{(d),(\underaccent{\tilde}{d}) \times (\tilde{d}), (d')} 
 \ = \ \sum_{\underaccent{\tilde}{d} = 0}^{d} \sum_{\tilde{d} = d'}^{\mathscr{D}}  q^*_{(d),(\underaccent{\tilde}{d}) \times (\tilde{d}), (d')}.
\] 
To get the constructed distribution to match the observed distribution, we will need to show that there exists $q^*_{d,d' \times d'', d'''}$ satisfying 
\begin{equation} 
\label{eqn:match} 
p_{(d),(d')} 
 \ = \ \sum_{\underaccent{\tilde}{d} = 0}^{d} \sum_{\tilde{d} = d'}^{\mathscr{D}}  q^*_{(d),(\underaccent{\tilde}{d}) \times (\tilde{d}), (d')}, 
 \end{equation} 
as well as ensuring that Assumption~\ref{asm:orthog}(b) holds for the contructed distribution.   For each $R_{d, d'} \ne \emptyset$, choose a point 
$r_{d, d'} \in R_{d, d'}$.  Define $(\varepsilon^*_s, \varepsilon^*_t) | x_s, x_t$ to be the discrete distribution on the support points $(r_{d, d'}, r_{d'', d'''})$, 
$\Pr( (\varepsilon^*_s, \varepsilon^*_t)= (r_{d, d'}, r_{d'', d'''}) | x_s, x_t ) = q^*_{d,d' \times d'', d'''}$.  So the marginal is 
$$\Pr(\varepsilon^*_s = r_{(d), (d')} | x_s, x_t ) \ = \  \sum_{\underaccent{\tilde}{d} = 0}^{\mathscr{D}} \sum_{\tilde{d} = \underaccent{\tilde}{d}}^{\mathscr{D}} 
q^*_{(d),(d') \times (\tilde{d}),(\underaccent{\tilde}{d})}.$$   The marginal for $\varepsilon^*_t | x_s, x_t $ is similar.  To ensure that 
Assumption~\ref{asm:orthog}(b) is satisfied we will need for the marginals to match, for $d \ge d'$, 
\begin{equation} 
\label{eqn:asm1b} 
0 \ = \ \sum_{ \underaccent{\tilde}{d} \le \tilde{d} } 
\left( q^*_{(\tilde{d}),(\underaccent{\tilde}{d})  \times (d),(d')} - q^*_{(d),(d') \times (\tilde{d}),(\underaccent{\tilde}{d})} 
\right).   
 \end{equation} 
In addition to equations (\ref{eqn:match}) and (\ref{eqn:asm1b}), the nonegativity inequalities $q^*_{d,d' \times d'', d'''} \ge 0$ must hold.  Let $p$ denote 
the vector of joint probabilities, 
$p = (p_{(\mathscr{D}),(\mathscr{D})}, p_{(\mathscr{D}),(\mathscr{D}-1)}, \ldots )^\prime$.  Let $q^*$ be the vector of probabilities 
$q^*_{(d),(d') \times (d''),(d''')}$ (where $d \ge d'$ and $d'' \ge d'''$), 
$q^* = (q^*_{(\mathscr{D}),(\mathscr{D}) \times (\mathscr{D}),(\mathscr{D}) }, \ldots)^\prime$.  And let $Q_s$ be the matrix with elements in $\{0,1\}$ 
such that equation (\ref{eqn:match}) can be restated as $p = Q_s q^*$, and let 
 $Q_p$ be the matrix with elements in $\{-1,0,1\}$ such that  equation (\ref{eqn:asm1b}) can be restated as $0 = Q_p q^*$.  

Our goal then can be summarized as showing that $\exists q^* \ge 0$ such that: (A) $p = Q_s q^*$; and (B) $0 = Q_p q^*$.  Let $z$ be a  $( \mathscr{D}+1)^2$-dimensional vector conformable with $p$, $z = (z_{(\mathscr{D}),(\mathscr{D})}, z_{(\mathscr{D}),(\mathscr{D}-1)}, \ldots )^\prime$.  Let $w$ be a 
$( \mathscr{D}+1)( \mathscr{D}+2)/2$-dimensional vector, $w = (\ldots, w_{(d), (d')}, \ldots)^\prime$.  Farkas' Lemma states that 
if 
\[ 
\left( \begin{array}{c} z \\ w \end{array} \right)^\prime  \left( \begin{array}{c} Q_s \\ Q_p \end{array} \right) \ge 0 \ \mbox{ implies}\ 
\left( \begin{array}{c} z \\ w \end{array} \right)^\prime  \left( \begin{array}{c} p \\ 0 \end{array} \right) = z^\prime p \ge 0, 
\] 
then 
$\exists q^* \ge 0$ satisfying (A) and (B) above.  

Each element $q^*_{(d),(d') \times (d''),(d''')}$ of $q^*$ appears in exactly one equation from constraints (A) and either zero or two (with positive and negative signs) from constraints (B).  In particular, elements of the form $q^*_{(d),(d) \times (d),(d)}$ (and $q^*_{(d),(d') \times (d),(d')}$ with $d>d'$) appear in (A) but not (B).  Hence, $z_{(d),(d)} \ge 0$ (and $z_{(d),(d')} \ge 0$ for $d>d'$).  Also, the conditional moment inequalities yield that for $d = 1, \ldots, \mathscr{D}$, 
$\Pr(y_s \in \{ (\mathscr{D}), \ldots, (d) \} | x_s, x_t) \ge \Pr(y_t \in \{ (\mathscr{D}), \ldots, (d) \} | x_s, x_t)$ which implies
\begin{equation}
\label{eqn:pmom} 
\sum_{\tilde{d}=d}^{\mathscr{D}} \sum_{\underaccent{\tilde}{d} = 0}^{d-1} p_{(\tilde{d}),(\underaccent{\tilde}{d})} 
\ \ge \ 
\sum_{\tilde{d}=d}^{\mathscr{D}} \sum_{\underaccent{\tilde}{d} = 0}^{d-1} p_{(\underaccent{\tilde}{d}),(\tilde{d})}  \ \ \ \mbox{for}  \ 
d = 1, \ldots, \mathscr{D}
\end{equation} 
Now, for constants $a_{d}$, 
\begin{eqnarray*} 
z^\prime p &=& \sum_{d=0}^{\mathscr{D}}   z_{(d),(d)} p_{(d),(d)} \ +  \  
\sum_{d > d'} \left( z_{(d),(d')} p_{(d),(d')} + z_{(d'),(d)} p_{(d'),(d)} \right) 
\\ 
&\ge & 
\sum_{d > d'} \left( z_{(d),(d')} p_{(d),(d')} + z_{(d'),(d)} p_{(d'),(d)} \right) 
\\ 
&=& 
\sum_{d=1}^{\mathscr{D}} a_{d}   \underbrace{ \left[  \sum_{\tilde{d}=d}^{\mathscr{D}} \sum_{\underaccent{\tilde}{d} = 0}^{d-1} p_{(\tilde{d}),(\underaccent{\tilde}{d})}  
- \sum_{\tilde{d}=d}^{\mathscr{D}} \sum_{\underaccent{\tilde}{d} = 0}^{d-1} p_{(\underaccent{\tilde}{d}),(\tilde{d})} \right]  }_{ \ge 0 \ \mbox{\scriptsize by} \ (\ref{eqn:pmom}) } 
\\ && 
+ 
\sum_{s=1}^{\mathscr{D}} \sum_{d=0}^{\mathscr{D}-s} \left\{  \left( z_{(d+s),(d)} - \left[ \sum_{d' = d+1}^{d+s} a_{d'} \right] \right) p_{(d+s),(d)}   
+  \left( z_{(d),(d+s)} - \left[ \sum_{d' = d+1}^{d+s} a_{d'} \right] \right) p_{(d),(d+s)}   \right\} 
\end{eqnarray*} 

So, given $\left( \begin{array}{c} z \\ w \end{array} \right)^\prime  \left( \begin{array}{c} Q_s \\ Q_p \end{array} \right) \ge 0$, we have $z^\prime p \ge 0$ 
if $\exists \  a_{\mathscr{D}}, \ldots, a_{1} \ge 0 $ such that 
$ - z_{(d),(d+s)} \le \sum_{d' = d+1}^{d+s} a_{d'} \le z_{(d+s),(d)}$ for $s = 1, \ldots, \mathscr{D}$,  $d = 0, \ldots, \mathscr{D}-s$.  

From examination of the constraints (A) and (B), $\left( \begin{array}{c} z \\ w \end{array} \right)^\prime  \left( \begin{array}{c} Q_s \\ Q_p \end{array} \right) \ge 0$ yields, for $d \ne d'$,  $z_{(d),(d')} + w_{(\tilde{d}),(d')} - w_{(d),(\underaccent{\tilde}{d})} \ge 0$ \ \ \ with $\tilde{d} \in \{ d', \ldots, \mathscr{D}\}, 
\underaccent{\tilde}{d} \in \{0, \ldots, d\}$.    For $d>d'$, let $\bar{a}_{d, d'+1}  = \max_{ \, \tilde{d} \ge  d', \, \underaccent{\tilde}{d} \le d} \ \ 
\left\{ w_{(d),(\underaccent{\tilde}{d})} -  w_{(\tilde{d}),(d')}  \right\}$ (and we will shorten $\bar{a}_{d,d}$ to $\bar{a}_{d}$).  For $d+1<d'$, let $\underaccent{\bar}{a}_{d',  d+1}  = \min_{ \, \tilde{d} \ge  d', \, \underaccent{\tilde}{d} \le d} \ \ 
\left\{w_{(\tilde{d}),(d')} - w_{(d),(\underaccent{\tilde}{d})}   \right\}$.  For $d+1 = d'$, let $\underaccent{\bar}{a}_{d'} = \max\left\{ 0 ,  \min_{ \, \tilde{d} \ge  d', \, \underaccent{\tilde}{d} \le d'-1} \ 
\left\{w_{(\tilde{d}),(d')} - w_{(d'-1),(\underaccent{\tilde}{d})}  \right\}  \right\}$. 
Then, for $d>d'$, $z_{(d),(d')} \ge \bar{a}_{d,  d'+1}$.  And, for $d<d'$, $-z_{(d),(d')} \le \underaccent{\bar}{a}_{d',  d+1}$, where we have imposed 
(\ref{eqn:pmom}) through the definition of $\underaccent{\bar}{a}_{d'}$.
%\begin{itemize} 
%\item[$\bullet$] $d>d'$: \ \ $z_{(d),(d')} + w_{(\tilde{d}),(d')} - w_{(d),(\underaccent{\tilde}{d})}$ \ \ \ for $\tilde{d} \in \{ d', \ldots, \mathscr{D}\}, 
%\underaccent{\tilde}{d} \in \{0, \ldots, d\}$
% 
%\item[$\bullet$] $d<d'$: \ \ $z_{(d),(d')} + w_{(\tilde{d}),(d')} - w_{(d),(\underaccent{\tilde}{d})}$ \ \ \ for $\tilde{d} \in \{ d', \ldots, \mathscr{D}\}, 
%\underaccent{\tilde}{d} \in \{0, \ldots, d\}$
%\end{itemize} 
So, to show $\exists \  a_{\mathscr{D}}, \ldots, a_{1} \ge 0 $ such that 
$ - z_{(d),(d+s)} \le \sum_{d' = d+1}^{d+s} a_{d'} \le z_{(d+s),(d)}$ for $s = 1, \ldots, \mathscr{D}$,  $d = 0, \ldots, \mathscr{D}-s$, 
 it will suffice to show $\exists \  a_{\mathscr{D}},  \ldots,  a_{1}$ such that  $\underaccent{\bar}{a}_{d} \le a_d \le \bar{a}_d$ for $d = 1, \ldots, \mathscr{D}$, and 
 $\underaccent{\bar}{a}_{d+s, d+1}  \le \sum_{d' = d+1}^{d+s} a_{d'} \le  \bar{a}_{d+s, d+1}$ for $s = 1, \ldots, \mathscr{D}$,  $d = 0, \ldots, \mathscr{D}-s$.
We can show this by proving that certain linear combinations of the lower bounds are smaller than certain linear combinations of the upper bounds.   Let $b_{\cdot}$ and $c_{\cdot}$ denote the coefficients in the linear combinations.   In particular, it is sufficient to show that 
\begin{equation} 
\label{eqn:bcineq}
\sum_{d = 1}^{\mathscr{D}} b_d \underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} b_{d', d} \underaccent{\bar}{a}_{d',  d} \le \sum_{d = 1}^{\mathscr{D}} c_d \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} c_{d', d} \bar{a}_{d',  d} 
\end{equation} 
where $b_{\cdot} \ge 0$ and $c_{\cdot} \ge 0$ and for $d = 1, \ldots, \mathscr{D}$, 
\begin{equation} 
\label{eqn:bc}
b_d + \sum_{1 \le d' \le d \le d'' \le \mathscr{D}} b_{d'', d'} \ = \ c_d + \sum_{1 \le d' \le d \le d'' \le \mathscr{D}} c_{d'', d'}.  
\end{equation} 

Before showing (\ref{eqn:bcineq}),  it will be useful to first note a fact about the coefficients $b_{\cdot}$ and $c_{\cdot}$.  Let $g_d = \min\{ b_d, c_d \}$ and $g_{d', d} = \min\{b_{d', d}, c_{d', d}\}$ for all $d'$, $d$.  
Equation (\ref{eqn:bcineq}) can be re-stated as
\begin{eqnarray*} 
\lefteqn{ 
\sum_{d = 1}^{\mathscr{D}} (b_d -g_d)\underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} (b_{d', d}-g_{d', d}) \underaccent{\bar}{a}_{d',  d} 
+ 
\left[ \sum_{d = 1}^{\mathscr{D}} g_d \underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}}  g_{d', d} \underaccent{\bar}{a}_{d',  d} \right] 
}
\\ 
&\le& \sum_{d = 1}^{\mathscr{D}} (c_d-g_d) \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} (c_{d', d}-g_{d', d})  \bar{a}_{d',  d} 
+ \left[ \sum_{d = 1}^{\mathscr{D}} g_d \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} g_{d', d} \bar{a}_{d',  d}  \right]. 
\end{eqnarray*} 
Since $\underaccent{\bar}{a}_{d} \le \bar{a}_{d}$ and $\underaccent{\bar}{a}_{d',  d}  \le \bar{a}_{d',  d} $
\[ 
 \sum_{d = 1}^{\mathscr{D}} g_d \underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}}  g_{d', d} \underaccent{\bar}{a}_{d',  d} \ \le \ 
  \sum_{d = 1}^{\mathscr{D}} g_d \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} g_{d', d} \bar{a}_{d',  d}. 
 \]
 So it will suffice to show 
 \[
 \sum_{d = 1}^{\mathscr{D}} (b_d -g_d)\underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} (b_{d', d}-g_{d', d}) \underaccent{\bar}{a}_{d',  d} 
\  \le \  
  \sum_{d = 1}^{\mathscr{D}} (c_d-g_d) \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} (c_{d', d}-g_{d', d})  \bar{a}_{d',  d}.
 \] 
 Or, more simply, in proving (\ref{eqn:bcineq}), we may assume either $b_d = 0$ or $c_d = 0$ holds for each $d$, and similarly $b_{d', d}=0$ or $c_{d', d}=0$ 
 for each $d'$, $d$, which is useful in the cases to be considered.  

Take the case where $ \underaccent{\bar}{a}_{d} >0$ $\forall d$, the argument for (\ref{eqn:bcineq}) follows. 
\begin{eqnarray*} 
\lefteqn{ 
\sum_{d = 1}^{\mathscr{D}} c_d \bar{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} c_{d', d} \bar{a}_{d',  d} 
} 
\\ 
& \ge & 
\sum_{d = 1}^{\mathscr{D}} c_d (w_{d,d} - w_{d-1, d-1}) 
 \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} c_{d', d}  (w_{d',d'} - w_{d-1, d-1}) 
 \\ 
& = & 
w_{\mathscr{D}, \mathscr{D}} \left[ c_{\mathscr{D}}  
\ + \ \sum_{d = 1}^{\mathscr{D}-1}   c_{\mathscr{D}, d} \right]  
\ + \  w_{\mathscr{D}-1, \mathscr{D}-1} \left[   ( c_{\mathscr{D}-1} - c_{\mathscr{D}}  ) \ + \ \sum_{d' = 1}^{\mathscr{D}-2} c_{\mathscr{D}-1, d'} \right] 
\\ 
&& 
 \ + \ \sum_{d = 2}^{\mathscr{D}-2}  w_{d,d}  \left[ ( c_d - c_{d+1} ) 
  \ + \ \sum_{d' = 1}^{d-1} c_{d, d'} \ - \ \sum_{d'=d+2}^{\mathscr{D}} c_{d', d+1} 
  \right] 
  \\ 
&& 
 \ + \ w_{1,1}  \left[ ( c_1 - c_{2} )  \ - \ \sum_{d'=3}^{\mathscr{D}} c_{d', 2}   \right]  
  \ + \ w_{0,0}  \left[ - c_1  \ - \ \sum_{d'=2}^{\mathscr{D}} c_{d', 1}   \right]  
   \\ 
& = & 
w_{\mathscr{D}, \mathscr{D}} \left[ b_{\mathscr{D}}  
\ + \ \sum_{d = 1}^{\mathscr{D}-1}   b_{\mathscr{D}, d} \right]  
\ + \  w_{\mathscr{D}-1, \mathscr{D}-1} \left[   ( b_{\mathscr{D}-1} - b_{\mathscr{D}}  ) \ + \ \sum_{d' = 1}^{\mathscr{D}-2} b_{\mathscr{D}-1, d'} \right] 
\\ 
&& 
 \ + \ \sum_{d = 2}^{\mathscr{D}-2}  w_{d,d}  \left[ ( b_d - b_{d+1} ) 
  \ + \ \sum_{d' = 1}^{d-1} b_{d, d'} \ - \ \sum_{d'=d+2}^{\mathscr{D}} b_{d', d+1} 
  \right] 
  \\ 
&& 
 \ + \ w_{1,1}  \left[ ( b_1 - b_{2} )  \ - \ \sum_{d'=3}^{\mathscr{D}} b_{d', 2}   \right]  
  \ + \ w_{0,0}  \left[ - b_1  \ - \ \sum_{d'=2}^{\mathscr{D}} b_{d', 1}   \right]  
  \\ 
& = & 
\sum_{d = 1}^{\mathscr{D}} b_d (w_{d,d} - w_{d-1, d-1}) 
 \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} b_{d', d}  (w_{d',d'} - w_{d-1, d-1})  
   \\ 
& \ge & 
 \sum_{d = 1}^{\mathscr{D}} c_d \underaccent{\bar}{a}_{d} \ + \ \sum_{d = 1}^{\mathscr{D}-1} \sum_{d' = d+1}^{\mathscr{D}} c_{d', d} \underaccent{\bar}{a}_{d',  d} 
\end{eqnarray*} 
where the second equality follows from differences of the equations given in (\ref{eqn:bc}).  

The other cases to check allow $ \underaccent{\bar}{a}_{d} =0$ for some $d$ and work similarly.  We have focused on the case where there is a strict covariate 
index ordering.  When the covariate index ordering is weak (includes some ties), the argument above simplifies.  Finally, having verified Farkas' Lemma, we can conclude that a constructed disturbance distribution exists that satisfies Assumption~\ref{asm:orthog} and generates a constructed outcome and covariate distribution that matches the observed distribution.  It follows that $\Theta_0 \subset \Theta_S$ and $\Theta_0$ is sharp. 
% 
\hfill $\Box$ 



\bigskip 
\bigskip 






\noindent 
\textbf{\textsc{Proof of Proposition~\ref{pro:ms}:}} 
 In binary choice, it will be useful note that 
when $\Pr(y_s = 1 | x_s, x_t) = \Pr(y_t = 1 | x_s, x_t)$ then $\Pr(y_s = 0 | x_s, x_t) = \Pr(y_t = 0 | x_s, x_t)$.  Take an arbitrary $\theta$.  Then, $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{0\}\}, \{\{1\}\}$, or $\{\{0\},\{1\}\}$.   In any of these cases, $H(x_s,x_t,\theta) = 0$, so $H(x_s,x_t,\theta) = 0$ $\forall \theta$ when $\Pr(y_s = 1 | x_s, x_t) = \Pr(y_t = 1 | x_s, x_t)$.  

Another useful finding is that for any $\theta$ such that  $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{0\},\{1\}\}$, $H(x_s,x_t,\theta)  
= \sum_{ D \in \overline{\mathbb{D}}(x_s, x_t, \theta)}
 E[{m}_D(y_s, y_t) \, | \, x_s, x_t ] = [\Pr(y_s = 0 | x_s, x_t) - \Pr(y_t = 0 | x_s, x_t) ] + [\Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t)] = 0$.  \\[.1ex]  

Take $\theta \in \Theta_0$ and show that $H(x_s,x_t,\theta) = H(x_s,x_t,\theta_0)$ a.s., so that $H(\theta) = H(\theta_0)$. 
  Cases:
\begin{itemize}
\item[(a)]  $\Delta x_1^\prime \theta >0$.  Then $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{1\}\}$, so $H(x_s,x_t,\theta) 
= \Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t)$.  Since $\theta \in \Theta_0$, $\Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) \ge 0$.  If $\Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) > 0$, then we must have by Proposition~\ref{pro:pin}, $\overline{\mathbb{D}}(x_s, x_t, \theta_0) = \{\{1\}\}$, so $H(x_s,x_t,\theta) 
= H(x_s,x_t,\theta_0)$.  If $\Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) = 0$, then as noted above $H(x_s,x_t,\theta) = 0 
= H(x_s,x_t,\theta_0)$.
\item[(b)]  $\Delta x_1^\prime \theta =0$.  Then, $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{0\},\{1\}\}$, so as noted above $H(x_s,x_t,\theta) = 0$.  Since 
$\theta \in \Theta_0$, we must have $\Pr(y_s = 0 | x_s, x_t) - \Pr(y_t = 0 | x_s, x_t) \ge 0$ and $\Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) \ge 0$.  So, 
$\Pr(y_s = 1 | x_s, x_t) = \Pr(y_t = 1 | x_s, x_t)$, and $H(x_s,x_t,\theta_0)=0$.  Hence, $H(x_s,x_t,\theta) 
= H(x_s,x_t,\theta_0)$.
\item[(c)]  $\Delta x_1^\prime \theta <0$.  Similar to case (a), $H(x_s,x_t,\theta) 
= H(x_s,x_t,\theta_0)$.
\end{itemize} 
So we have shown that $H(x_s,x_t,\theta) = H(x_s,x_t,\theta_0)$ a.s., and so $H(\theta) = H(\theta_0)$.\\[.1ex] 

Now take $\theta \not\in \Theta_0$ and show $H(\theta) < H(\theta_0)$.  Let $\mathcal{A} = \{ (x_s, x_t): 
E[{m}_D(y_s, y_t) \, | \, x_s, x_t ] \ < \ 0 \ \mbox{for some }\ D \in \overline{\mathbb{D}}(x_s, x_t, \theta) \}$.  Since $\theta \not\in \Theta_0$, $\Pr(\mathcal{A}) >0$.    Consider $(x_s, x_t) \in \mathcal{A}$. 
\begin{itemize}
\item[(a)] $\Delta x_1^\prime \theta >0$.  But $\Pr(y_s = 1 | x_s, x_t) < \Pr(y_t = 1 | x_s, x_t)$.  Then, $\Pr(y_s = 0 | x_s, x_t) > \Pr(y_t = 0 | x_s, x_t)$ and 
$\Delta x_1^\prime \theta_0 < 0$.  Hence, $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{1\}\}$ and $\overline{\mathbb{D}}(x_s, x_t, \theta_0) = \{\{0\}\}$, and 
$H(x_s,x_t,\theta)= \Pr(y_s =  1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) < 0 <  \Pr(y_s =  0| x_s, x_t) - \Pr(y_t = 0 | x_s, x_t)  = H(x_s,x_t,\theta_0)$.  
\item[(b)]  $\Delta x_1^\prime \theta =0$. Then, $\overline{\mathbb{D}}(x_s, x_t, \theta) = \{\{0\},\{1\}\}$, so $H(x_s,x_t,\theta) = 0$.  But since $(x_s, x_t) \in \mathcal{A}$,  $\Pr(y_s = 1 | x_s, x_t) \ne \Pr(y_t = 1 | x_s, x_t)$.  Suppose $\Pr(y_s = 1 | x_s, x_t) > \Pr(y_t = 1 | x_s, x_t)$, then 
$\Delta x_1^\prime \theta_0 > 0$, and $H(x_s,x_t,\theta_0) = \Pr(y_s = 1 | x_s, x_t) - \Pr(y_t = 1 | x_s, x_t) > 0 = H(x_s,x_t,\theta)$.  The argument 
holds similarly when $\Pr(y_s = 1 | x_s, x_t) < \Pr(y_t = 1 | x_s, x_t)$.  So $H(x_s,x_t,\theta_0) > H(x_s,x_t,\theta)$
\item[(c)]  $\Delta x_1^\prime \theta <0$.  Arguing similarly to (a), $H(x_s,x_t,\theta_0) > H(x_s,x_t,\theta)$.  
\end{itemize} 
So we have shown that $H(x_s,x_t,\theta_0) > H(x_s,x_t,\theta)$ for $(x_s, x_t) \in \mathcal{A}$.  For $(x_s, x_t) \not\in \mathcal{A}$, it is straightforward to 
show $H(x_s,x_t,\theta) = H(x_s,x_t,\theta_0)$.  Hence, $H(\theta_0) - H(\theta) = E[ {\bf 1}\{ (x_s, x_t) \in \mathcal{A} \} (H(x_s,x_t,\theta_0) - H(x_s,x_t,\theta))] >0$.  The result follows.
%
\hfill $\Box$ 






\end{document} 



