%\documentclass{slides}
\documentclass[landscape]{seminar}
%\documentclass[twoside]{seminar}
%\usepackage{fancyhdr}
\usepackage{amssymb}
\usepackage{latexsym}
\input epsf
\usepackage{psfig}
%\usepackage[hyperindex,pdfmark]{hyperref}
%\usepackage{fullpage}
%\usepackage{multicol}
%\usepackage[pdftex]{graphicx}
%\usepackage{hyperref}
\pagestyle{plain}
\newtheorem{exam}{Example}[subsubsection]
\newtheorem{prob}{Problem}[subsubsection]
\newtheorem{definition}{Definition}[subsubsection]
\newtheorem{remark}{Remark}[subsubsection]
\newtheorem{prop}{Proposition}[subsubsection]
\newtheorem{lemma}{Lemma}[subsubsection]
\newtheorem{lem}{Lemma}[subsubsection]
\newtheorem{thm}{Theorem}[subsubsection]
\newtheorem{cor}{Corollary}[subsubsection]
\newtheorem{alg}{Algorithm}[subsubsection]
\def\half{\frac{1}{2}}
\def\Rm {\mathbb{R}^m}
\def\Rn {\mathbb{R}^n}
\def\RR {\mathbb{R}}
\def\pols {\mathbb{P}}
\newcommand{\D}[2]{[{\cal D}#1(#2)]}
\newcommand{\inner}[2]{{\langle #1 , #2 \rangle}}
\newcommand{\infpgms}[2]{\inf\,\Bigl\{#1 \bigm| #2\Bigr\}}
\newcommand{\suppgms}[2]{\sup\,\Bigl\{#1 \bigm| #2\Bigr\}}
\newcommand{\infpgm}[3]{\mbox{#1~~~~~}\inf\,\Bigl\{#2 \bigm| #3\Bigr\}}
\newcommand{\suppgm}[3]{\mbox{#1~~~~~}\sup\,\Bigl\{#2 \bigm| #3\Bigr\}}
\newcommand{\minpgm}[3]{\mbox{#1~~~~~}\min\,\Bigl\{#2 \bigm| #3\Bigr\}}
\newcommand{\maxpgm}[3]{\mbox{#1~~~~~}\max\,\Bigl\{#2 \bigm| #3\Bigr\}}
\newcommand{\minpgms}[2]{\min\,\Bigl\{#1 \bigm| #2\Bigr\}}
\newcommand{\maxpgms}[2]{\max\,\Bigl\{#1 \bigm| #2\Bigr\}}
\newcommand{\req}[1]{(\ref{#1})}
\newcommand{\svec}{{\rm svec\,}}
\newcommand{\hMat}{{\rm hMat\,}}
\newcommand{\vsMat}{{\rm vsMat\,}}
\newcommand{\kmat}{{\rm Mat\,}}
\newcommand{\sMat}{{\rm sMat\,}}
\newcommand{\dsvec}{{\rm dsvec\,}}
\newcommand{\Hmat}{{\rm Hmat\,}}
\newcommand{\vSmat}{{\rm vSmat\,}}
\newcommand{\Smat}{{\rm Smat\,}}
\newcommand{\sdiag}{{\rm sdiag\,}}
\newcommand{\sn}{{\cal S}^n }
\def\sdp {{\bf SDP}}
\def\sqp {{\bf SQP}}
\def\nep {{\bf NEP}}
\def\Lag {{\cal L}}
\def\lkp {\lambda^{(k+1)}}
\def\lk {\lambda^{(k)}}
\def\sqqp {{\bf SQQP}}
\def\xk {{x}^{(k)}}
\def\xkp {{x}^{(k+1)}}
\def\df {:=}
\def\pA{{\widetilde A}}
\def\pZ{{\widetilde Z}}
\def\pX{{\widetilde X}}
\def\pdx{{\widetilde{dx}}}
\def\pdz{{\widetilde{dz}}}
\def\pfd{{\widetilde{f_d}}}
\def\pfc{{\widetilde{f_c}}}
\def\pdy{{\widetilde{dy}}}
\def\pfp{{\widetilde{f_p}}}
\def\V{\mathbb V}
\def\L{{\cal L}}                %The Lagrangian
\def\F{{\cal F}}                %The feasible region
\def\Z{{\cal Z}}
\def\X{{\cal X}}
\def\I{{\cal I}}
\def\P{{\cal P}}
\def\RE{\mathbb R}
\def\Mmn {\RE^{m\times n}}
\def\SE {\mathbb S}
\def\Snp {\SE^n_+}
\def\Snpp {\SE^n_{++}}
\newcommand{\pD}[3]{[{\cal D}_#2#1(#3)]}
\newcommand{\Di}[3]{[{\cal D}^#2#1(#3)]}
\newcommand{\Mn}{{\cal M}^n }
\newcommand{\Sn}{{\cal S}^n }
\newcommand{\hn}{{\cal H}^n }
\newcommand{\p}{{\cal P} }
\newcommand{\g}{{\cal G} }
\newcommand{\kvec}{{\rm vec\,}}
\newcommand{\adj}{{\rm adj\,}}
\newcommand{\intr}{{\rm int\,}}
\newcommand{\trace}{{\rm trace\,}}
\newcommand{\relint}{{\rm relint\,}}
\newcommand{\rank}{{\rm rank\,}}
\newcommand{\cone}{{\rm cone\,}}
\newcommand{\tr}{{\rm trace\,}}
\newcommand{\diag}{{\rm diag\,}}
\newcommand{\Diag}{{\rm Diag\,}}
\newcommand{\Se}{{\mathcal S}_e }
\newcommand{\Sd}{{\mathcal S}_d }
\newcommand{\Sc}{{\mathcal S}_C }
\newcommand{\Sh}{{\mathcal S}_H }
\newcommand{\snn}{{\mathcal S}_{n-1} }
\newcommand{\GG}{{\mathcal G} }
\newcommand{\KK}{{\mathcal K} }
\newcommand{\LL}{{\mathcal L} }
\newcommand{\DD}{{\mathcal D} }
\newcommand{\BB}{{\mathcal B} }
\newcommand{\PP}{{\mathcal P} }
\newcommand{\TT}{{\mathcal T} }
\newcommand{\FF}{{\mathcal F} }
\newcommand{\HH}{{\mathcal H} }
\newcommand{\EE}{{\mathcal E} }
\newcommand{\NN}{{\mathcal N} }
\newcommand{\RRc}{{\mathcal R} }
\newcommand{\UU}{{\mathcal U} }
\newcommand\T{{\mathcal T}}
\newcommand{\bt}{ \begin{tabular} }
\newcommand{\et}{ \end{tabular} }
\newcommand\A{{\mathcal A}}
\newcommand\E{{\mathcal E}}
\newcommand{\bpr}{{\bf Proof.} \hspace{1 em}}
\newcounter{count}
%\newcommand{\beq}{\begin{equation}}
%\newcommand{\eeq}{\end{equation}}
%\newcommand{\epr}{\\ \hspace*{4.5in}  $\Box$}
\renewcommand{\theequation}{\thesection.\thecount}
%\newcommand{\beq}{\begin{equation}}
\newcommand{\beq}{\addtocounter{count}{1} \begin{equation}}
\newcommand{\bet}{\addtocounter{count}{1} \begin{table}}
\newcommand{\eeq}{\end{equation}}
%\newcommand{\beqr}{\begin{eqnarray}}
\newcommand{\beqr}{\addtocounter{count}{1} \begin{eqnarray}}
\newcommand{\addc}{\addtocounter{count}{1} }
\newcommand{\bs}{\setcounter{count}{0} \section}
%\newcommand{\bs}{\section}

\newcommand{\QED}{\hfill ~\rule[-1pt] {8pt}{8pt}\par\medskip ~~}
\newcommand{\epr}{\QED}
\newcommand{\arrow}{{\rm arrow\,}}
\newcommand{\Arrow}{{\rm Arrow\,}}
\newcommand{\trian}{{\rm trian\,}}
\newcommand{\Trian}{{\rm Trian\,}}
\newcommand{\BoDiag}{{\rm B^0Diag\,}}
\newcommand{\OoDiag}{{\rm O^0Diag\,}}
\newcommand{\bodiag}{{\rm b^0diag\,}}
\newcommand{\oodiag}{{\rm o^0diag\,}}
\begin{document}
\bibliographystyle{plain}
\title{
\begin{figure}
\psfig{file=UWlogori.ps,height=10mm}
\end{figure}
Introduction to Semidefinite Programming: Motivation and Duality}
\author{
%\href{http://orion.math.uwaterloo.ca/~hwolkowi}
{Henry Wolkowicz}\thanks{
Department of Combinatorics \& Optimization,
          University of Waterloo
                       }
       }

\maketitle
%\tableofcontents
\newpage

\begin{slide}{}

{Outline}

\begin{enumerate}
\item[$\bullet$]
Basic Properties and Notation
\item[$\bullet$]
Examples
\item[$\bullet$]
Duality

\end{enumerate}


\end{slide}
\begin{slide}{}

\bs{Basic Properties and Notation}
\label{sect:basicpropnot}

Basic linear {\bf Semidefinite Programming}
 looks just like Linear Programming
\[ {\bf (PSDP)}
\begin{array}{cccc}
    p^*=  & \max &\tr CX & (\left< C,X \right>) \\
 &  \mbox{s.t.} & {\cal A}X = b & \mbox{(linear)}\\
  && X \succeq 0,~~(X \in \p)& \mbox{(nonneg)}
    \end{array}
\]

\[
{\cal A}(\alpha X_1+\beta X_2)=
\alpha {\cal A}(X_1)+ \beta {\cal A}(X_2), \quad \forall X_i, \mbox{ and scalars }
\alpha, \beta
\]

\end{slide}
\begin{slide}{}
${\cal S}^n$ := space of $n \times n$ real symmetric matrices,~~($A=A^T$)
(Hermitian?)


$\preceq$ denotes the L{\"{o}}wner partial order\\
$A\preceq B$ if $B-A \succeq 0$ (positive semidefinite)\\

\hrulefill

For $A \in {\cal S}^n$:\\
A is positive semidefinite (positive definite),
({\tiny{denoted  $A \succeq 0$ ($A \succ 0$),}})\\ 
if $x^TAx \geq 0 (>0), ~\forall x\neq 0$.

\hrulefill

For
$M, n \times n$, $Mx=\lambda x$ (eigenvector-eigenvalue equation, simplest linear
operation is multiplication by scalar), equivalently, $\det (M-\lambda I)=0$.



\end{slide}
\begin{slide}{}


TFAE:
\begin{enumerate}
\item
$A \succeq 0 \quad (A \succ 0)$
\item
the eigenvalues $\lambda(A) \geq 0
\quad (\lambda(A) > 0)$
\item  $A = R^TR$, for some $R$ (Cholesky factorization if $R$ is upper
triangular - Gram matrix representation
of $n$ columns of $R$, $a_{ij}=u_i^Tu_j$, $R$ nonsingular if $A \succ 0$)
\item
$A= \sum_i  \alpha_i v_iv_i^T$ for some nonnegative $\alpha_i$ (sum of rank ones,
$A \succ 0$??)
\item
all principal minors (determinants of principal submatrices) are nonnegative
(positive for $A \succ 0$, leading princ. minors??)
( principal submatrix: delete some rows
and corresp. cols.; history-Sylvester?)
\item
$S=(S^{\frac 12})^2$,  for some $S^{\frac 12} \succeq 0 \quad (\succ 0)$
\end{enumerate}


\end{slide}
\begin{slide}{}


For a general square matrix 
$M, n \times n$,  define\\
 $\trace M= \sum_i M_{ii}$
Then 
 \[\trace M= \sum_i \lambda_{i}
 \quad \mbox{(trace is sum of eigenvalues)}
\]


\end{slide}
\begin{slide}{}

\begin{thm}
(Interlacing/Majorization Eigenvalues, Fan, 1954) 
Let $H,\bar{H}$ be an $n \times n$ Hermitian matrices  of the form
\[
H=\pmatrix{  H_{11}&H_{12}\cr
             H_{21}&H_{22}
}, \qquad
\bar{H}=\pmatrix{  H_{11}&0\cr
             0 &H_{22}
},
\]
with $H_{11}$ size $l \times l$. Then
\[
  (\lambda (H_{11}) ,\lambda (H_{22})) =\lambda (\bar{H}) \prec_M \lambda (H),
 \quad \mbox{on } \Re^n,
\]
where $\prec_M$ denotes majorization, i.e. $\lambda \prec_M \mu$ \underline{if}
$\sum_{i=1}^k \lambda_{[i]} \leq
 \sum_{i=1}^k \mu_{[i]}
$, with equality for $k=n$.  ([.] denotes nonincreasing order). In particular,
$\lambda_j(H) \geq \lambda_j(H_{11}), \quad 
\lambda_{n-j+1}(H) \leq \lambda_{l-j+1}(H_{11})$.
\epr
\end{thm}



\end{slide}
\begin{slide}{}
\subsubsection{Further Properties and Definitions}
\label{sect:furthpropdefs}

\begin{enumerate}
\item[$\bullet$]
For matrices $P,Q$ compatible for multiplication, 
\[  \trace PQ = \trace QP.  \]
\item[$\bullet$]
Every symmetric matrix $S$ has an orthogonal decomposition $S=Q\Lambda
Q^T$, where $\Lambda$ is diagonal with the eigenvalues of $S$ on the
diagonal, and $Q$ has orthonormal columns consisting of eigenvectors of $S$.
\item[$\bullet$]
$X \succeq 0 \Rightarrow  P^TXP \succeq 0 \quad \mbox{(congruence)}$

\end{enumerate}



\end{slide}
\begin{slide}{}
Further properties of $A \succeq 0$:\\
$\bullet$
$A= \sum^r_{i=1}  \alpha_i v_iv_i^T$ for minimum number of nonnegative $\alpha_i$
characterizes $\rank A$;\\
$\bullet$
$\diag(A) \geq 0$; and $a_{ii}=0$ implies i-th and i-th col are 0;\\
$\bullet$
$\trace A \geq 0$ and $\trace A = 0$ iff $A=0$;\\
$\bullet$
 $\p \subset \Sn$ is a closed convex cone, i.e.
$\p \subset  \p + \p; \alpha \p \subset  \p, \forall \alpha \geq 0$; 
$\bullet$
$A,B \succeq 0$ implies $\trace AB \geq 0$ with equality iff $AB=0$ (proof?).
$\bullet$
geometry of $\p$ in $\Sn$ equipped with $\trace$ inner product: all vectors within 45
degrees of identity matrix;\\
$\bullet$
$\p$ is a {\em self-polar} cone (proof??), $\p = \p^*$, where
\[
C^* = \left\{ \phi \in \Re^n : \left<\phi,x\right> \geq 0, \forall x \in C \right\}.
\]


\end{slide}
\begin{slide}{}

\begin{lem} {\bf (Schur Complement)}
Let $A= \pmatrix{  B & C^T \cr C & D }$ be a symmetric matrix with the
block $B\succ 0$. Then
\[ A \succ 0 ~  (\succeq 0)  \mbox{  iff  } D-CB^{-1}C^T \succ 0 ~  (\succeq 0).
\]
\end{lem}

\end{slide}
\begin{slide}{}

\begin{thm}
(Perron-Frobenius) If an $n \times n$ matrix has nonnegative entries then it has a
nonegative real eigenvalue $\lambda$ which has maximum absolute value among all
eigenvalues. This eigenvalue $\lambda$ has a nonnegative real eigenvector. If, in
addition, the matrix has no block-triangular decomposition (is irreducible - does not
contain a $k \times (n-k)$ block of 0-s disjoint from the diagonal), then $\lambda$
has multiplicity 1 and the corresponding eigenvector is positive.
\epr
\end{thm}

\end{slide}
\begin{slide}{}



The linear operator ${\cal A}$  and its adjoint:
\[ 
{\cal A} :{\cal S}^n \rightarrow \Rm, \qquad
{\cal A}^* : \Rm \rightarrow {\cal S}^n
\]
\[
\begin{array}{c}
  \mbox{for given}~ A_i \in {\cal S}^n, i=1,\ldots m:\\
({\cal A}X)_i =\tr (A_iX), \quad {\cal A}^*y= \sum_{i=1}^m y_iA_i
\end{array}
\]
where
\[
\left<{\cal A}X,y\right>=\left<X, {\cal A}^*y\right>, \quad \forall X,y
\]

$\p$ - cone of positive semidefinite matrices\\
replaces\\
$\Rn_+$ - nonnegative orthant


\end{slide}
\begin{slide}{}

$\trace CX = \left<C,X\right>=\sum_{i=1}^n \sum_{j=1}^n C_{ij} X_{ij}$

SDP is equivalent to:

\[ {\bf (PSDP)}
\begin{array}{cccc}
 p^*=  & \max & \sum_{i=1}^n \sum_{j=1}^n C_{ij} X_{ij}&\\
 &  \mbox{s.t.} & \left<A_i,X\right>=b_i, & i=1,\ldots m \\
  && X \succeq 0,~~(X \in \p)
    \end{array}
\]

\end{slide}
\begin{slide}{}


\subsubsection{Duality (like LP)}
\label{sect:duality}
payoff function, player $Y$ to player $X$ (Lagrangian)
\[ L(X,y) :=  \tr (CX) +y^T(b-{\cal A}X)
\]

Optimal (worst case) strategy for player $X$:
\[p^* = 
      \max_{X \succeq 0 }  \min_{y }  L(X,y) 
\]
Using the {\em hidden constraint} $b-{\cal A}X=0$, recovers primal problem
(PSDP).

\end{slide}
\begin{slide}{}
\[ 
\begin{array}{rcl}
L(X,y) &=&  \tr (CX) +y^T(b-{\cal A}X)\\
       &=&  b^Ty + \tr \left(C -{\cal A}^*y \right) X
\end{array}
\]

~\\
~\\
adjoint operator,~~ ${\cal A}^*y= \sum_i y_i A_i $
\[
\left<{\cal A}^*y,X \right>= \left<y,{\cal A}X \right>, ~~~ \forall X,y
\]


\[p^* = 
       \max_{X \succeq 0 } \min_{y }  L(X,y) 
\leq d^*:=\min_y \max_{X \succeq 0} L(X,y) 
\]

For dual, use
{\em hidden constraint} $g(y) := C-{\cal A}^*y \preceq 0$
\end{slide}
\begin{slide}{}

\[p^* = 
       \max_{X \succeq 0 } \min_{y }  L(X,y) 
\leq d^*:=\min_y \max_{X \succeq 0} L(X,y) 
\]
dual obtained from optimal strategy of competing player
Y;\\
use {\em hidden constraint} $g(y) = C-{\cal A}^*y \preceq 0$
\[ {\bf (DSDP)} \qquad
\begin{array}{ccc}
    d^*=& \min &b^Ty \\
 &  \mbox{s.t.} & {\cal A}^*y \succeq  C \\
    \end{array}
\]

for the primal
\[ {\bf (PSDP)} \qquad
\begin{array}{ccc}
    p^*=  & \max &\tr CX \\
 &  \mbox{s.t.} & {\cal A}X = b\\
  && X \succeq 0
    \end{array}
\]

\end{slide}
\begin{slide}{}

\subsubsection{Weak Duality - Optimality}
\label{sect:weakduality}
\begin{prop} (Weak Duality)
If $X$ feas. in (PSDP), $y$ feas. in (DSDP), 
$Z=C-{\cal A}^*y \succeq 0$ is slack variable, then
\[ \mbox{duality gap } := b^Ty-\tr CX  = \trace XZ \geq 0.
\]
\end{prop}
\bpr (Direct - using 
$\trace ZX = \tr X^{1/2}   X^{1/2}Z = \tr X^{1/2} Z  X^{1/2} \geq 0$)
\begin{eqnarray*}
\tr CX - b^Ty &=&\tr ({\cal A}^*y-Z)X - b^Ty\\
&=&\tr y^T{\cal A}X-  \trace ZX - b^Ty\\
&=&\tr y^T\left({\cal A}X-b\right) -\trace ZX =-\trace ZX,
\end{eqnarray*}
\epr

\end{slide}
\begin{slide}{}

Characterization of optimality \\
(in the case of a zero duality gap at optimality and attainment)\\
 for the
   dual pair $X,y$~~(slack $Z\succeq 0$)
 \[ 
\begin{array}{cc} 
    {\cal A}^*y -Z = C  & \mbox{dual feasibility}\\
    AX = b & \mbox{primal feasibility}\\
    ZX  = 0 & \mbox{complementary slackness  (equiv. } \trace ZX=0)
\end{array}
\]
\[
    ZX = \mu I ~~~~~ \mbox{perturbed}
\]

Forms the basis for:\\ ~~\\
interior point methods\\
(primal simplex method,
dual simplex method)

\end{slide}
\begin{slide}{}

\subsubsection{Preliminary Examples}
\label{sect:prelexamples}
\begin{exam}
{\bf Minimizing the Maximum Eigenvalue}
Arises in e.g. stability of differential equations
\begin{enumerate}
\item[$\bullet$]
The mathematical problem:
\begin{enumerate}
\item[$\diamond$]
given $A(x)$ depending linearly on vector $x$
\item[$\diamond$]
Find $x$ to minimize the maximum eigenvalue of $A(x)$
\end{enumerate}
\item[$\bullet$]
SDP Model:
\begin{enumerate}
\item[$\diamond$]
the largest eigenvalue $\lambda_{\max}(A(x)) \leq \alpha \qquad$
\underline{iff}
$\lambda_{\max}(A(x)-\alpha I) \leq 0 \qquad$ \underline{iff}
$ \qquad A(x)-\alpha I \preceq 0$.
\item[$\diamond$]
The DSDP (in dual form) is:
\[  \max -\alpha \qquad \mbox{s.t.} \quad  A(x)- \alpha I \preceq 0. \]

\end{enumerate}
\end{enumerate}

\end{exam}


\end{slide}
\begin{slide}{}
{\bf HOW DOES SDP arise from quadratic approximations?}
\[\mbox{Let} \quad q_i(y)=\frac 12 y^TQ_iy+y^Tb_i + c_i,~y\in \Rn \]
\[ {\bf (QQP)} \qquad
\left\{ \begin{array}{ccc}
    q^*=  & \min &q_0(y) \\
 &  \mbox{s.t.} & q_i(y)=0\\
     &&  i=1,\ldots m
    \end{array}   \right.
\]
$\bullet$ Lagrangian (quadratic, linear, constant in $y$ terms):
\[ 
  L(y,x) = \frac 12 y^T (Q_0 -\sum_{i=1}^m x_iQ_i)y
 +y^T(b_0 -\sum_{i=1}^m x_ib_i) + (c_0 -\sum_{i=1}^m x_ic_i)
\]
\[\mbox{$\bullet$ Primal-Dual pair:} \quad 
    q^*=\min_y \max_x L(y,x) \geq d^* = \max_x \min_y L(y,x)
\]
\[
\mbox{$\bullet$ homogenize (add $y_0$): } \quad 
 y_0y^T(b_0 -\sum_{i=1}^m x_ib_i), ~~ y_0^2=1.  \]
\end{slide}
\begin{slide}{}
\[
\begin{array}{cccc}
  d^* &=& \max_x \min_y &L(y,x)\\
  &=& \max_x \min\limits_{y_0^2=1}& \frac 12 y^T (Q_0 -\sum_{i=1}^m
x_iQ_i)y
                            ~~~(+ty_0^2) \\
 &&&   +y_0y^T(b_0 -\sum_{i=1}^m x_ib_i)\\
   &&&+ (c_0 -\sum_{i=1}^m x_ic_i)
                            ~~~(-t)
\end{array}
\]
{\em hidden semidefinite constraint} yields SDP constraint
\[
(\A: \RR^{m+1} \rightarrow {\cal S}_{n+1})\qquad
B-\A \left( \begin{array}{c}
      t \\ x
   \end{array}  \right)
 \succeq 0.
\]
\[
B=\left( \begin{array}{cc}
      0 & b_0^T \\ b_0 &Q_0
   \end{array}  \right), \quad
\A \left( \begin{array}{c}
      t \\ x
   \end{array}  \right)
   = \left[ \begin{array}{cc}
        -t &  \sum_{i=1}^m x_ib_i^T \\
      \sum_{i=1}^m x_ib_i  & \sum_{i=1}^m x_i Q_i
        \end{array}   \right]
\]

\end{slide}
\begin{slide}{}
The dual program is equivalent to the SDP (with $c_0=0$)
\[ {\bf (D)} \qquad
\begin{array}{ccc}
    d^*=  & \mbox{sup} & -\sum_{i=1}^m x_ic_i -t \\
 &  \mbox{s.t.} & \A\left( \begin{array}{c}
      t \\ x
   \end{array}  \right)
 \preceq B\\
  && x \in \Rm, t \in \RR
    \end{array}
\]
As in linear programming, the dual is obtained from the optimal
strategy of the competing player:
\[ {\bf (DD)} \qquad
\begin{array}{ccc}
    d^*=& \inf &\tr BU \\
 &  \mbox{s.t.} & \A^*U = \left( \begin{array}{c}
      -1 \\ -c
   \end{array}  \right) \\
  && U \succeq 0.
    \end{array}
\]



\end{slide}
\begin{slide}{}

\begin{exam}
{\bf Pseudoconvex (Nonlinear) Optimization Problem}

\[ {\bf (PCP)} \qquad
\begin{array}{ccc}
    d^*=& \min &\frac {(c^Tx)^2}{d^Tx} \\
 &  \mbox{s.t.} & A x +b \geq  0,
    \end{array}
\]
(where $Ax+b \geq 0 \Rightarrow d^T x > 0$) is equivalent to
\[ 
\begin{array}{ccc}
    d^*=& \min & t \\
 &  \mbox{s.t.} & 
   \left[ \begin{array}{ccc}
           \Diag(A x +b) & 0 & 0 \\
              0  &  t  &  c^Tx  \\
          0  &   c^T x  &  d^T x 
    \end{array}  \right]   \succeq 0
    \end{array}
\]
\end{exam}


\end{slide}
\begin{slide}{}

$\bullet$
{\bf \em Alternate Name - Linear Matrix Inequality} 
\[\qquad g(y) = C-{\cal A}^*y \preceq 0\] 
adjoint operator $\qquad {\cal A}^*y= \sum_i y_i A_i, \quad A_i=A_i^T $
~~\\
$\bullet$
$g(y)$ is a {\bf convex constraint}, i.e.
\[ g(\lambda x + (1-\lambda)y) \preceq \lambda g(x) + (1-\lambda)g(y),
\quad \forall 0 \leq \lambda \leq 1, \forall x,y \in \Rm
\]
equivalently, for every $P \succeq 0$,
the real valued function $f_P(x) := \trace Pg(x)$ is a convex 
function, i.e.
\[ f_P(\lambda x + (1-\lambda)y) \leq \lambda f_P(x) + (1-\lambda)f_P(y),
\quad \forall 0 \leq \lambda \leq 1, \forall x,y \in \Rm
\]

\end{slide}
\begin{slide}{}

$\bullet$
{\em feasible set} 
$\left\{ x \in \Rm : g(x) \preceq 0 \right\}$ is convex set

$\bullet$
{\em optimum point} is on the boundary ($g(x)$ is singular)

$\bullet$
{\em boundary of feasible set} is {\em not} smooth (piecewise smooth,
algebraic surface)

\begin{figure}[h]
\centering\psfig{file=laurpol.ps,height=50mm}
\end{figure}



\end{slide}
\begin{slide}{}


\subsection{Introduction to Convex Cones and the L\"owner Partial Order}
\label{sect:introCC}
\begin{definition}
Let $\alpha \in \RR$ and $S,T \subset \Rn$. Then
$\alpha S = \{y:y=\alpha s, \mbox{ for some } s \in S\}$ and
$S+T = \{y:y=s+t, \mbox{ for some } s \in S, t \in T\}$
\end{definition}
\begin{definition}
 $\KK \subset \mathbb{R}^n$ is a \underline{cone} if 
$\alpha \KK \subset K, \quad  \forall   \alpha  > 0$.
\end{definition}

\begin{definition}
the cone $\KK$ is a \underline{convex cone} if $\KK + \KK \subset \KK$.
\end{definition}

\begin{definition}
A cone $\KK$ is a \underline{pointed cone} if $\KK \cap (-\KK)=\{ 0 \}$.
\end{definition}

\begin{definition}
A cone $\KK \subset \mathbb{R}^n$ is a \underline{proper cone} if it is closed, 
pointed, and convex and has nonempty interior.
\end{definition}
 

\end{slide}
\begin{slide}{}

{\bf Examples of Cones}
\begin{exam}
\underline{open half line:} $\{x\in\mathbb{R}: x > 0 \}$.
\end{exam}

\begin{exam}
\underline{closed half line:} $\{x\in\mathbb{R}: x\geq 0 \}$.
\end{exam}
 
\begin{exam}
\underline{psd matrices, $\p$:} 
$\{X\in \Sn: X\succeq 0 \}$.
\end{exam}

\begin{exam}
\underline{Lorentz cone}\\
 (ice-cream cone, second-order cone):\\
$L^m =\{ x=(x_1,\ldots,x_m)\in \Rm: x_m \geq 
  \sqrt{x_1^2+ \ldots +x_m^2}$.
\end{exam}


\end{slide}
\begin{slide}{}

And, an important closed convex cone is:
\begin{definition}
\underline{Polar (dual, conjugate)} of set ${\mathcal S}$: 
${\mathcal S}^+ := \{ z: \left< x, z\right> \geq 0, \forall  
 x\in {\mathcal S} \} \quad (=S^*)$.
\end{definition}

\begin{exam}
nonnegative orthant, psd cone $\p$,  and Lorentz cone $L^m$,
 are all self-polar, i.e. $\KK = \KK^+$.
\end{exam}


\end{slide}
\begin{slide}{}

\begin{definition}
\underline{Direct sum of two cones} 
$\KK \oplus \LL := \{ (k~ l) :  k \in \KK,  l \in \LL \}$.
\end{definition}
 
\begin{exam}
Let $L$ denote the half line in $\RR$, 
then $\underbrace{L\oplus L \cdots \oplus L}_n$ is the $n$-dimension 
nonnegative orthant.
\end{exam}



\end{slide}
\begin{slide}{}


Suppose $\KK$, $\KK_1$, $\KK_2$ are proper cones, then:
\begin{itemize}
\item
$\KK^+$ is proper.
\item
\( (\KK^+)^+=\KK \) .
\item
\( (\KK_1  \cap \KK_2)^+=\overline{\KK_1^++\KK_2^+} \).
\item
\( (\KK_1+\KK_2)^+=\KK_1^+\cap \KK_2^+ \).
\item
\( \KK_1\oplus \KK_2 \) is proper.
\item
\( ({\mathcal K}_1\oplus {\mathcal K}_2)^+
={\mathcal K}_1^+ \oplus {\mathcal K}_2^+ \). 
\end{itemize}

Useful Lemma, e.g. to Prove Farkas' Lemma
\begin{lem}
\[
\KK \mbox{ is a closed convex cone }  \iff  \KK = \KK^{++}.
\]
\end{lem}

\end{slide}
\begin{slide}{}

Partial Orders:
\begin{definition}
${ x{\succeq}_{\mathcal K}  y} 
\mbox{ (respectively }{ x\succ_{\mathcal K}  y)}$ 
if $ x- y \in \KK  \mbox{ (respectively }x- y \in \intr \KK)$.

\end{definition}
 
\begin{remark}
If $\KK$ is a pointed, convex cone, then ``$\succeq_\KK$'' is a 
\underline{(linear)
partial order} ( reflexive, transitive, and antisymmetric):
\begin{itemize}
\item
$0 \in {\mathcal K}$ $\Rightarrow  x\succeq_{\mathcal K}  x$
(reflexive);
\item
${\mathcal K}$ is convex $\Rightarrow \mbox{if }  x\succeq_{\mathcal K}  y
\mbox{ and } y\succeq_{\mathcal K}  z, 
                \mbox{ then }  x\succeq_{\mathcal K}  z$
(transitive);
\item
${\mathcal K}$ is pointed $\Rightarrow  \mbox{if } x\succeq_{\mathcal K}  y  
\mbox{ and } y\succeq_{\mathcal K}  x, \mbox{ then }  x= y$ (antisymmetric);  
\item
{${\mathcal K}$ is convex cone $\Rightarrow$}
$\mbox{if } a,b \geq 0,  u\succeq_{\mathcal K}  x
\mbox{ and } v\succeq_{\mathcal K}  y, 
                \mbox{ then }  au+bv \succeq_{\mathcal K}  ax+by$
(linear - homogeneous, additive);
\end{itemize}
\end{remark}



\end{slide}
\begin{slide}{}

\subsection{Convex Cone Program}
\label{sect:ccp}
Let:  $\KK,\LL$ be convex cones;\\
$f$ real valued convex function, $g$ is $\KK$-convex, i.e.
$g(\alpha u + (1-\alpha) v) \preceq_\KK 
   \alpha g(u) +  (1-\alpha) g(v), \forall 0\leq \alpha \leq 1,
      \forall u,v$
\[
(CP)\qquad \begin{array}{cc}
\mu^*=\min &  f(x) \\
{\mbox{s.t.}} & g(x) \preceq_\KK 0\\
    &  x \succeq_\LL 0
\end{array}
\]
with Lagrangian dual (weak duality)
\[
(DCP)\qquad \begin{array}{cc}
\mu^* \geq \nu^*=\max\limits_{y \in \KK^+} \min\limits_{x \in \LL} 
            & f(x) +\left<y,g(x)\right>
\end{array}
\]

\end{slide}
\begin{slide}{}

\begin{exam}
If $f(x)=\left<c,x \right>, g(x)=b-{\mathcal A} x$, then we have a linear cone
programming problem, (LCP). The dual
\[
(DLCP)\qquad \begin{array}{rclc}
\mu^* \geq \nu^*&=&\max\limits_{y \in \KK^+} \min\limits_{x \in \LL} 
            & \left<c,x\right> +\left<y,b-{\mathcal A} x \right> \\
               &=&\max\limits_{y \in \KK^+} \min\limits_{x \in \LL} 
            & \left<c-{\mathcal A}^*y,x\right> +\left<y,b \right> \\
      &=&\max\limits_{\stackrel{y \in \KK^+}{c-{\mathcal A}^*y \in \LL^+}}
        \min\limits_{x \in \LL} 
            & \left<c-{\mathcal A}^*y,x\right> +\left<y,b \right> \\
\end{array}
\]
reduces to the elegant (LP or SDP type) form
\[
(DLCP)\qquad \begin{array}{cc}
\mu^* \geq \nu^*=\max & \left<y,b \right> \\
              \mbox{s.t.} &  {\mathcal A}^*y \preceq_{\LL^+} c\\
       &  y \succeq_{\KK^+} 0
\end{array}
\]
\end{exam}

\end{slide}
\begin{slide}{}

\[
(LCP)\qquad \begin{array}{cc}
\mu^*=\min &  \left<c,x\right> \\
{\mbox{s.t.}} & {\cal A}x \succeq_\KK b\\
    &  x \succeq_\LL 0
\end{array}
\]

\[
(DLCP)\qquad \begin{array}{cc}
\mu^* \geq \nu^*=\max & \left<y,b \right> \\
              \mbox{s.t.} &  {\mathcal A}^*y \preceq_{\LL^+} c\\
       &  y \succeq_{\KK^+} 0
\end{array}
\]
\begin{thm}
Suppose that Slater's CQ holds for (LCP), i.e. $\exists \hat{x}$ such that
$ {\cal A}\hat{x} \succ_\KK b,\quad \hat{x} \succ_\LL 0$. Then {\em strong duality} holds for
(LCP),(DLCP), i.e. there exists $y^*$ feasible for (DLCP) such that
\[\mu^*=\nu^*=\left<y^*,b\right>.\]
\end{thm}

\end{slide}
\begin{slide}{}

\begin{exam}

If the primal is
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \sup &x_2 \\
 &  \mbox{s.t.} & 
\left[ \begin{array}{ccc}
       x_2 & 0 & 0\\
       0 & x_1 & x_2\\
       0 & x_2 & 0
       \end{array}  \right]
\preceq 
   \left[ \begin{array}{ccc}    
       1 & 0 & 0\\
       0 & 0 & 0\\
       0 & 0 & 0
       \end{array}  \right] 
    \end{array}
\]
$S^f = cone \left\{ 
   \left[ \begin{array}{ccc}
       1 & 0 & 0\\
       0 & 0 & 0\\
       0 & 0 & 0
       \end{array}  \right]
\right\} \quad \mbox{(minimal cone containing feasible set)}
$

\newpage

Then the dual is
\[ {\bf (D)}
\begin{array}{ccc}
    d^*=& \inf & U_{11}\\
 &  \mbox{s.t.} & U_{22}=0  \\
 &   & U_{11}+2U_{23}=1  \\
  && U \succeq 0.
    \end{array}
\]
Then $p^*=0 < d^*=1.$

--------------------------\\
(But regularized dual (below) has $ U \succeq_{(S^f)^+} 0,$ i.e. only constraint
is $U_{11} \geq 0.$ So new dual optimal value is 0. (Attained.) )

\end{exam}

\end{slide}
\begin{slide}{}

\subsection{Optimality Conditions without constraint qualifications}
\subsubsection{Facial Structure}
\begin{definition}
$\FF$ is a {\em face} of a convex cone $\KK$
(denoted $\FF \lhd \KK$) if $\FF \subset \KK$ and
\[
\frac 12 (x_1 + x_2) \in \FF \Rightarrow x_1,x_2 \in \FF, \qquad
\forall x_1,x_2 \in \KK
\]
\end{definition}

\begin{definition}
A face $\FF \lhd \KK$ is called {\em exposed} if
\[
\FF = \KK \cap \phi^{\perp}, \qquad \mbox{for some } \phi \in \KK^+
\]
\end{definition}
\begin{definition}
A face $\FF \lhd \p$ is called {\em projectionally exposed} if
\[
\FF = Q \p Q^t \qquad \mbox{for some matrix } Q
\]
\end{definition}

\end{slide}
\begin{slide}{}

\subsubsection{Weak Duality}
The weak duality: $ c^T x \geq b^T y$ holds in Cone-LP:
\[  c^T x -b^T y =  c^T x-(A x)^T y =  x^T ( c-A^T y)= x^T  s \geq 0  \] 

\subsubsection{Extended Farkas' Lemma}
\begin{thm}
\textbf{(Seperation Theorem)} If $K \subseteq \mathbb{R}^n$ is closed and convex, $b \in \mathbb{R}^n$, $b \not\in K$, then $\exists  a \in \mathbb{R}^n$, such that $\forall  x \in K:  a^T  b<0$ and $ a^T x\geq0$ .
\end{thm}

\begin{lem}
\label{eq:farkaslem}
\textbf{(Farkas' Lemma) }Let $\KK  \subseteq \mathbb{R}^n$ be a closed, 
convex cone, $A \in \mathbb{R}^{m \times n}, A({\mathcal K})$ closed,  
and $b \in \mathbb{R}^m $. Then:
\[
\left\{\exists  x \succeq_{\mathcal K} 0 : A x=b \right\}
\iff 
\left\{ A^T y \succeq_{{\mathcal K}^+} 0 \Rightarrow
b^T y \geq 0 \right\}.
\]
\end{lem}




\end{slide}
\begin{slide}{}


\bpr
(Necessity:)
Suppose $\exists  x \in {\mathcal K} : A x=b$. And suppose
$A^T y \succeq_{{\mathcal K}^+ } 0$.
Then $b^T y=(A x)^T y= x^T(A^T y)\geq 0$,  since $x \succeq_K 0$.

(Sufficiency:)
Suppose that the RHS of (\ref{eq:farkaslem}) holds but
the LHS fails, i.e.
$b \not\in A({\mathcal K})$. Then separate, i.e.
$\exists  y \in \mathbb{R}_m$ such that  $b^T y < 0$, and $\forall  
x \in {\mathcal K}, (A x)^Ty \geq 0$. 
Therefore, $A^Ty \in K^+$ which implies $b^Ty \geq 0$, contradiction.
\epr




\end{slide}
\begin{slide}{}

Cone programming: $S$ a closed convex cone induces a
linear partial order (\cite{bw2})
\[ 
\min \{f(x) : g(x) \preceq_S 0, ~  x \in \Omega \}  \]
feasible set $A$

--------------------------\\
rewrite using minimal face $g(A) \subset -S^f$
\[ 
\min \{f(x) : g(x) \preceq_{S^f} 0, ~  x \in \Omega^f\}  \]
where
$\Omega^f 
= \Omega \cap g^{-1} (S^f-S^f)  
= \Omega \cap g^{-1} (S^f-S)  $

--------------------------\\
optimality conditions for some $\Lambda \in (S^f)^+$:
\[ f(x) + \Lambda g(x) \geq \mu^*, ~\forall x \in
\Omega^f \]





\end{slide}
\begin{slide}{}
Special case of linear cone programming, $\KK,\LL$ closed
convex cones (\cite{w11}):

primal:
\[  \mu^*=
\min \{cx : \A x \succeq_\KK 0, ~  x \succeq_\LL 0 \}  \]

minimal cones $\KK^f, \LL^f$

dual:
\[ \mu^*= \nu^*=
\max \{by : \A^*y \preceq_{(\LL^f)^+} b, ~  
          y \succeq_{(\KK^f)^+} 0 \}  \]

\end{slide}
\begin{slide}{}

Linear Programming  ($\A:\Rm \rightarrow \Rn$)

$\Rn_+$ is a closed convex cone

\[ {\bf (P)}
\begin{array}{cccc}
    p^*=  & \mbox{sup} &c^Tx \\
 &  \mbox{s.t.} & \A x \preceq b&(b-\A x \in \Rn_+)\\
  && x \in \Rm
    \end{array}
\]
Lagrangian (payoff function):\\ 
$L(x,U)=c^Tx+\left< U,b-\A x \right>$
\[p^*=\max_x \min_{U \succeq 0} L(x,U) \]
(the constraint $U \succeq 0$ is needed to recover\\ 
the hidden constraint $\A x \preceq b$.)
\end{slide}
\begin{slide}{}
The dual is obtained from the optimal strategy of the competing player
\[p^*\leq d^*=\min_{U \succeq 0} \max_x L(x,U) =
   \left< U,b\right> +x^T(c-\A^*U)
\]
The hidden constraint $c-\A^*U=0$ yields the dual
\[ {\bf (D)}
\begin{array}{ccc}
    d^*=& \inf &\tr bU \\
 &  \mbox{s.t.} & \A^*U = c \\
  && U \succeq 0.
    \end{array}
\]
for the primal
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \mbox{sup} &c^Tx \\
 &  \mbox{s.t.} & \A x \preceq b\\
  && x \in \Rm
    \end{array}
\]
\end{slide}
\begin{slide}{}
If Slater's condition fails for the primal LP, 
then there are an infinite number of
different dual programs.

The implicit equality constraints are:
  \[  \A_ex = b_e  \]
where $\A = \left[ \begin{array}{c}   \A_l \\ \A_e  \end{array} \right]$

\[ {\bf (D)}
\begin{array}{ccc}
    d^*=& \inf &\tr bU \\
 &  \mbox{s.t.} & \A_l^*U_l
            +\A_e^*U_e = c \\
  && U\in {\cal U}\\
 &&    
           \left\{U:U \succeq 0 \right\} \subset {\cal U} \\
       &&  {\cal U} \subset \left\{U:U_l \succeq 0, U_e ~\mbox{free} \right\}.
    \end{array}
\]
for the equivalent primal program
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \mbox{sup} &c^Tx \\
 &  \mbox{s.t.} & \A_lx \preceq b_l\\
 &  & \A_ex = b_e\\
  && x \in \Rm
    \end{array}
\]
\end{slide}
\begin{slide}{}
\begin{center}
{\bf DUALITY THEOREM}
\end{center}
\begin{enumerate}
\item
If one of the problems is inconsistent, then the other is inconsistent
or unbounded.
\item
{\bf WEAK DUALITY}\\
Let the two problems be consistent, and let $x^0$ be a feasible solution
for P and $U^0$ be a feasible solution for D. Then
\[  c^Tx^0 \leq \left< b,U^0\right>. \]
\item
{\bf STRONG DUALITY}\\
If both P and D are consistent, then they have optimal solutions and
their optimal values are equal.
\newpage
\item
{\bf COMPLEMENTARY SLACKNESS}\\
Let $x^0$ and $U^0$ be feasible solutions of P and D, respectively.
Then $x^0$ and $U^0$ are optimal if and only if
\[  \left< U^0,(b-\A x^0)\right>=0. \]
if and only if
\[   U^0 \circ (b-\A x^0)=0. \]
\item
{\bf SADDLE POINT}\\
The vectors $x^0, U^0$ are optimal solutions
of P and D, respectively, if and only if $(x^0,U^0)$ is a saddle
point of the Lagrangian $L(x,U)$ for all (x,U),
\[  L(x,U^0) \leq L(x^0,U^0) \leq  L(x^0,U)    \]
 and then
\[  L(x^0,U^0) = c^Tx^0= \left<b,U^0\right>.   \]
\end{enumerate}

\end{slide}
\begin{slide}{}
Characterization of optimality for the\\
   dual pair $x,U$
 \[ 
\begin{array}{cc} 
    \A x \preceq b & \mbox{primal feasibility}\\
~\\
    \A^*U = c  & \mbox{dual feasibility}\\
~\\
    U \circ (\A x -b) = 0e & \mbox{complementary slackness}
\end{array}
\]
\[
    U \circ (\A x -b) = \mu e ~~~~~ \mbox{perturbed}
\]

Forms the basis for:\\ ~~\\
primal simplex method\\
dual simplex method\\
interior point methods

\end{slide}
\begin{slide}{}
\begin{enumerate}
\item[]
What is a proper duality theory?\\
\item[]
Do duality gaps occur in practice?
\item[]
Are there an infinite number of duals if Slater's condition fails?
\end{enumerate}
\end{slide}
\begin{slide}{}
\begin{center}
{\bf Facial Structure for $\p$}
\end{center}
the cone $T
\subset \KK$ is a {\em face} of the cone $\KK$, denoted $T \lhd \KK$, if
\[ x,y \in \KK,~ x+y \in T \Rightarrow x,y \in T. \]
Each face, $\KK \lhd \p$, is
characterized by a subspace, $S \subset \Rn.$
\[
\KK = \{ X \in \p : \NN (X) \supset S \}.
\]
\[
\relint \KK = \{ X \in \p : \NN (X) = S \}.
\]
The complementary face of $\KK$ is $\KK^c = \KK^{\perp} \cap \p$ 
\[
\KK^c = \{ X \in \p : \NN (X) \supset S^{\perp} \}.
\]
\[
\relint \KK^c = \{ X \in \p : \NN (X) = S^{\perp} \}.
\]
\end{slide}
\begin{slide}{}
the face $\KK$ (respectively, $\KK^c$)
is determined by the supporting hyperplane corresponding to any $X \in
\relint \KK^c$ (respectively, $\relint \KK$); 

and 
\[XY=0, \quad \forall X \in \KK, \forall Y \in \KK^c
\]
\end{slide}
\begin{slide}{}

An SDP:
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \mbox{max} &c^Tx \\
 &  \mbox{s.t.} & \A x  \preceq_{\p} b \\
  && x \in \Rm.
    \end{array}
\]

\[
\mbox{Optimality conditions:}\qquad
c \in \A^* (\p^+)  ~~({\rm closed?})
\]

dual program

\[ {\bf (D)}
\begin{array}{ccc}
    p^*=& \min &\tr bU \\
 &  \mbox{s.t.} & \A^*U = c \\
  && U \succeq_{\p^+} 0.
    \end{array}
\]
\[
\mbox{Optimality conditions:}\qquad
b \in {\cal R}(\A)+ \p^+  ~~({\rm closed?})
\]


\end{slide}
\begin{slide}{}
The {\em minimal cone} of P is defined as
\[
 \p^f = \cap \{ \mbox{faces of~} {\cal P} \mbox{~containing~} (b- \A(F))
\}.
\]
Therefore, an equivalent program is
the {\em regularized P program}
\[ {\bf (RP)}
\begin{array}{ccc}
    p^*=  & \mbox{max} &c^Tx \\
 &  \mbox{s.t.} & \A x  \preceq_{\p^f} b \\
  && x \in \Rm.
    \end{array}
\]
there exists $x$ such that $b-\A x \in \relint \p^f.$
(generalized Slater's constraint qualification)
Strong duality pair is RP and
\[ {\bf (DRP)}
\begin{array}{ccc}
    p^*=& \min &\tr bU \\
 &  \mbox{s.t.} & \A^*U = c \\
  && U \succeq_{(\p^f)^+} 0.
    \end{array}
\]
\end{slide}
\begin{slide}{}
Find $\p^f$? Properties?

{\bf LEMMA 1}\\
Suppose $\p^f \lhd \KK \lhd \p.$ Then the system
\[
 \A^*U=0, U \succeq_{\KK^+} 0, \tr Ub=0
\]
is consistent only if
\[
\mbox{the minimal cone}~ \p^f \subset \{U\}^{\perp} \cap \KK.
\]
{\bf PROOF}\\ 
Since $\tr U(\A x-b)=0$, for all $x$, we get $\A(F)-b \subset
{U}^{\perp},$
i.e. $\p^f \subset \{U\}^{\perp}.$
~~\\
\hspace*{4.5in}  $\Box$

\end{slide}
\begin{slide}{}

{\bf LEMMA 2} (surprising)\\
Suppose that $0\neq \KK \lhd \p$ (proper face). Then
\[   \p^+ + \KK^{\perp}
   = cl(\p^+ + span~\KK^c)
 ~\mbox{is always closed}.
\]
But
 \[   \p^+ + span ~\KK~ \mbox{is never closed}.
\]

~~\\
\hspace*{4.5in}  $\Box$


\end{slide}
\begin{slide}{}

Define: $\p_0:=\p$ and
\[ \UU_{1} := \{ U \succeq_{(\p_0)^+} 0 :
 \A^*U=0,  \tr Ub=0 \}
\]
Choose $U_1 \in \UU_{1} \cap \relint \UU_{1}^f $ (if $U_1=0$ - {\bf STOP})
\[ \p_1 := \UU_{1}^c = \{ U_1 \}^{\perp} \cap \p_0 \lhd \p_0
\]
\[ {\bf (RP_1)}
\begin{array}{ccc}
    p^*=  & \mbox{max} &c^Tx \\
 &  \mbox{s.t.} & \A x  \preceq_{\p_1} b \\
  && x \in \Rm.
    \end{array}
\]
We get: 
$\qquad p^*\leq d_1^* \leq d^*$
\[ {\bf (DRP_1)}
\begin{array}{ccc}
    d_1^*=& \min &\tr bU \\
 &  \mbox{s.t.} & \A^*U = c \\
  && U \succeq_{(\p_1)^+} 0.
    \end{array}
\]

\end{slide}
\begin{slide}{}

$(\p_1)^+=(\p \cap \p_1)^+ = \p + (\p_1)^{\perp}$
\[ {\bf (ELSD_1)} \qquad
\begin{array}{ccc}
    d_1^*=& \min &\tr b(U+(W+W^T)) \\
 &  \mbox{s.t.} & \A^*(U+(W+W^T)) = c \\
 &  & \A^*U_1= 0, \tr U_1 b =0 \\
   &&U \succeq 0,~\left[ \begin{array}{cc}    
       I & W^T \\
       W & U_1
       \end{array}  \right] 
   \succeq 0.
    \end{array}
\]

\end{slide}
\begin{slide}{}

We used
\[
\begin{array}{cc}
\p_1^{\perp}=& \left\{ (W+W^T) : \A^*U_1 = 0, \tr U_1 b = 0, \right. \\
& ~~ \left. \left[ \begin{array}{cc}  I & W^T \\
                      W & U_1  \end{array}  \right ]   \succeq 0
                             \right\}
\end{array}
\]
and
\[  \begin{array}{ccc}
U_1 \succeq WW^T  &\mbox{iff} &
  \left[ \begin{array}{cc}  I & W^T \\
                      W & U_1  \end{array}  \right ]   \succeq 0\\
& \mbox{implies}& W=U_1H, \mbox{for some matrix H}
\end{array}
\]

\end{slide}
\begin{slide}{}

\begin{eqnarray*} 
\UU_2& := &\{ U \succeq_{(\p_1)^+} 0 : \A^*U=0,  \tr Ub=0 \}\\
       &= &\{ U + Z:  \A^*(U+Z)=0,  \tr Ub=0, \\
        &&  ~~~U \succeq_{(\p_0)^+}, Z \in (\p_1)^{\perp}\}
\end{eqnarray*}
Choose $U_2 \in \UU_2 \cap \relint \UU_2^f $ (if $U_2=0$ - {\bf STOP})
\[ \p_2 := \UU_2^c = \{ U_2 \}^{\perp} \cap \p_1 \lhd \p_1
\]

\end{slide}
\begin{slide}{}

\[ {\bf (RP_2)}
\begin{array}{ccc}
    p^*=  & \mbox{max} &c^Tx \\
 &  \mbox{s.t.} & \A x  \preceq_{\p_2} b \\
  && x \in \Rm.
    \end{array}
\]
$p^*\leq d_2^*\leq d_1^* \leq d^*$
\[ {\bf (DRP_2)}
\begin{array}{ccc}
    d_2^*=& \min &\tr bU \\
 &  \mbox{s.t.} & \A^*U = c \\
  && U \succeq_{(\p_2)^+} 0.
    \end{array}
\]


$(\p_2)^+=(\p  \cap \p_2)^+ = \p + (\p_2)^{\perp}$

\end{slide}
\begin{slide}{}

\[ {\bf (ELSD_2)} \qquad
\begin{array}{ccc}
    d_1^*=& \min &\tr b(U+(W+W^T)) \\
 &  \mbox{s.t.} & \A^*(U+(W+W^T)) = c \\
 &  & \A^*U_1= 0, \tr U_1 b =0 \\
 &  & \A^*(U_2+(W_1+W_1^T))= 0, \\
    &&\tr (U_2+(W_1+W_1^T)) b =0 \\
   &&U \succeq 0,~\left[ \begin{array}{cc}    
       I & W_1^T \\
       W_1 & U_1
       \end{array}  \right] 
   \succeq 0\\
   &&\left[ \begin{array}{cc}    
       I & W^T \\
       W & U_2
       \end{array}  \right] 
   \succeq 0.
    \end{array}
\]
\end{slide}
\begin{slide}{}
HOMOGENIZATION (Alternate view of optim. cond.)
 \[ {\bf (HP)} \qquad
\begin{array}{cccc}
    0=& \max &c^Tx+t(-p^*)\\
 &  \mbox{~subject to~} & \A x + t(-b) +Z = 0 \\
  && w \in \KK=\Rm  \otimes \RR_+ \otimes \p\\
    \end{array}
\]
This defines the objective, constraints, and variables:
 \[
\begin{array}{cccc}
 (=\left< a,w \right> )\\
(\BB w =0)\\
           \left(w= \left( \begin{array}{c}  x\\t\\Z
    \end{array} \right)\right)
    \end{array}
\]
the feasible set is
\[  F_H = \NN (\BB) \cap \KK, \quad
(\BB w=0, w \in \KK ~\mbox{implies~} \left< a,w \right> \leq 0)
\]
\end{slide}
\begin{slide}{}
Optimality conditions:
\[
a =  \left( \begin{array}{c} c\\-p\\0 \end{array} \right) \in
    -( \NN (\BB) \cap \KK)^+.
\]
\[
 \left( \begin{array}{c} -c\\p\\0 \end{array} \right) \in
    \overline{\RRc (\BB^*) +  \KK^+},
\]
WCQ - Weakest Constraint Qualification:
\begin{center}
CLOSURE HOLDS
\end{center}
\end{slide}
\begin{slide}{}
Conditions for closure:
\begin{enumerate}
\item[]
If $C,D$ are closed convex sets and the intersection of their recession cones
is empty, then $D-C$ is closed.
\item[]
$cone (F_H-\KK)$ is the whole space
\item[]
(Slater's) \[ \exists \hat{x} \in F ~\mbox{such that}~ \A \hat{x} \prec
b.
\]
\end{enumerate}
FIX: Find sets, $T$,
to add to attain the closure. Equivalently, find sets, $C,~C^+=T$,
to intersect with $\KK$ to attain the closure since
\[
    (\NN (\A) \cap  (\KK \cap C))^+ =
    \overline{\RRc (\A^*) +  \KK^+ + C^+}.
\]
\end{slide}
\begin{slide}{}


\end{slide}
\begin{slide}{}

\subsection{Logarithmic Barrier Function and the Central Path}
\label{sect:logbarrier}
References:\\
$\bullet$Fiacco and McCormick \cite{MR91d:90089} 1968, barrier techniques (SUMT)\\
$\bullet$Nesterov and Nemirovskii \cite{int:Nesterov5} 1990, self-concordant barriers\\
$\bullet$ $\log\det(X)$ is a strictly concave 
barrier on pos def cone (e.g. Bellman 1960, \cite{Bell:60})

~~\\

family of strictly convex primal-dual (e.g. \cite{Le:94}) pairs,
 parameterized by the scalar $\mu>0$,
\begin{eqnarray*}
    \min \left\{\trace CX-\mu\log\det(X) : \A(X)=b,X\succ 0
          \right\},\\
    \max \left\{b y+\mu\log\det(Z) : \A^*(y)+Z=C,Z\succ 0\right\}.
\end{eqnarray*}

\end{slide}
\begin{slide}{}

  \begin{eqnarray*}
    \infpgm{$(P_\mu)$}{\inner{C}{X}-\mu\log\det(X)}{\A(X)=b},\\
    \suppgm{$(D_\mu)$}{\inner{b}{y}+\mu\log\det(Z)}{\A^*(y)+Z=C}.
  \end{eqnarray*}
log-barrier pair parametrized by $\mu$\\
objective functions are strictly convex/concave\\
under Slater's (strict feasibility), there exists unique (positive
definite) solutions $X_\mu, y_\mu, Z_\mu$



\end{slide}
\begin{slide}{}


Lagrangians:
\begin{eqnarray*}
  \L_P(X,y) &:=& \inner{C}{X}-\mu\log\det(X) + \inner{y}{b-\A(X)},\\
  \L_D(X,y,Z) &:=& \inner{b}{y}+\mu\log\det(Z) + \inner{X}{C-Z-\A^*(y)}.
\end{eqnarray*}
stationarity:
\[
  0 = \nabla{\L_P}{(X,y)} = 
  \pmatrix{
    C - \mu X^{-1} - \A^*(y)\cr
    b-\A(X)
  }
\]
add $Z:=\mu X^{-1}$
\[
  0 = \nabla{\L_D}{(X,y,Z)} = 
  \pmatrix{
    C - Z - \A^*(y)\cr
    b-\A(X)\cr
    \mu Z^{-1} - X
  }
\]


\end{slide}
\begin{slide}{}

\[
\mbox{optimality conditions} \quad \left\{
  \begin{array}{c}
           \begin{array}{rcl}
             \A^*(y)+Z &=& C,\\
             \A(X) &=& b,\\
           \end{array} \\
     \begin{array}{|c|}
   \hline\\
    \star \star \quad   Z = \mu X^{-1}.\\
   ~~\\
   \hline
     \end{array}
  \end{array}
\right.
\]


\end{slide}
\begin{slide}{}

For all $\mu > 0$, the set of unique solutions,
denoted $(X_\mu,y_\mu,Z_\mu)$ is called the \textit{primal-dual
  central path}.  


Note complementarity:
\begin{displaymath}
  \inner{Z_\mu}{X_\mu} = \inner{\mu X^{-1}_\mu}{X_\mu} = \mu n.
\end{displaymath}

\end{slide}
\begin{slide}{}

    {\bf Central Path }
\begin{center}
%\begin{figure}[htbp]
%  \begin{center}
%   \epsfig{file=cp.eps,height=1.5in}
    \psfig{file=cp.eps,height=1.5in}
%    \caption{Central Path }
%    \label{fig:cp}
%  \end{center}
%\end{figure}
   \end{center}


\end{slide}
\begin{slide}{}



\subsubsection{Nonlinearity and Smoothing}
\label{sec:nonlinearity}
\[
    Z - \mu X^{-1}=0 \quad  \mbox{is highly nonlinear}
  \]

Replace with (multiplication by $X$)
\beq
  \label{optmu_cs_lin}
  ZX = \mu I.
\eeq
to reduce the nonlinearity of the system with the aim of accelerating
Newton-like methods.



\end{slide}
\begin{slide}{}


Consider Newton-like method to find $v^*$ such that $F(v^*)=0$.  \\
(From e.g. \cite{DennSch:83} Theorems 5.2.1 and 10.2.1) 
radius ($r$) for quadratic convergence
is bounded by a measure of \textit{relative nonlinearity} of $F$
\begin{displaymath}
  r \propto \frac{k}{\beta \gamma}.
\end{displaymath}
$k$ small constant that depends on method (Newton or Gauss-Newton),
\begin{displaymath}
  \|{\F^{\prime}}{(v^*)}^{-1}\| \le \beta, \quad \mbox{(bound)}
\end{displaymath}
$\gamma$ is Lipschitz continuity constant for $\F^{\prime}{(v)}$ near $v^*$. 

Equivalently: (condition number of Jacobian inverse)
\begin{displaymath}
  r \propto k \frac{\sigma_{\min} \F^{\prime}{(v^*)}}{\sigma_{\max} 
          \F^{\prime}{(v^*)}}.
\end{displaymath}



\end{slide}
\begin{slide}{}

for $F(X,Z):=Z-\mu X^{-1}$: we have
\begin{displaymath}
  \F^{\prime}{(X,Z)}(dX,dZ) = dZ + \mu X^{-1}dX X^{-1}.
\end{displaymath}
$\bullet$ optimal solution $X^*$ is almost always rank deficient\\
$\bullet$ condition number is arbitrarily large\\
$\bullet$ proven radius of quadratic convergence is exceedingly small

{\bf BUT} for $F(X,Z)=ZX-\mu I$:
\begin{displaymath}
  \F^{\prime}{(X,Z)}(dX,dZ) = ZdX + dZX,
\end{displaymath}
$\bullet$ $\sigma_{\max}$ can be bounded by $||Z|| + ||X||$\\
(Jacobian is full-rank from standard assumptions)\\
$\bullet$ we obtain a nonzero radius of quadratic convergence.





\end{slide}
\begin{slide}{}

Summary - optimality conditions (look like for LP):
  \begin{eqnarray}
  \label{S_optmu}
    \A^*(y)+Z &=& C,\label{S_optmu_df}\\
    \A(X) &=& b,\label{S_optmu_pf}\\
    ZX &=& \mu I\label{S_optmu_cs}.
  \end{eqnarray}


\end{slide}
\begin{slide}{}

\subsection{Differences with LP}
\label{sect:diffLP}
\begin{enumerate}
\item
Duality gaps can exist for SDP 
in the absence of strictly feasible solutions (Slater's constraint
qualification, CQ).

\item
 Strict complementarity can fail at the optimum.

\item
 The perturbed optimality conditions involve nondiagonal matrices and
map between different spaces\\
 (overdetermined -does it require symmetrization?).
\end{enumerate}

\end{slide}
\begin{slide}{}

\subsubsection{Duality Gaps}
\begin{exam}
If the primal is
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \max &x_2 \\
 &  \mbox{s. t.} & 
\left[ \begin{array}{ccc}
       x_2 & 0 & 0\\
       0 & x_1 & x_2\\
       0 & x_2 & 0
       \end{array}  \right]
\preceq 
   \left[ \begin{array}{ccc}    
       1 & 0 & 0\\
       0 & 0 & 0\\
       0 & 0 & 0
       \end{array}  \right] 
    \end{array}
\]
then the dual is
\[ {\bf (D)}
\begin{array}{ccc}
    d^*=& \min &\tr U_{11}\\
 &  \mbox{s. t.} & U_{22}=0,\quad U_{11}+2U_{23}=1  \\
  && U \succeq 0.
    \end{array}
\]
Then $p^*=0 < d^*=1.$

\end{exam}

\end{slide}
\begin{slide}{}


\subsubsection{Strict Complementarity}

\begin{exam}
If the primal is
\[ {\bf (P)}
\begin{array}{ccc}
    p^*=  & \max &x_1 \\
 &  \mbox{s. t.} & 
\left[ \begin{array}{ccc}
       x_1 & x_3 & x_2\\
       x_3 & x_2 &  0 \\
       x_2 &  0  & x_3
       \end{array}  \right]
\preceq 
   \left[ \begin{array}{ccc}    
       0 & 0 & 0\\
       0 & 0 & 0\\
       0 & 0 & 1
       \end{array}  \right] 
    \end{array}
\]
then the dual is
\[ {\bf (D)}
\begin{array}{ccc}
    d^*=& \min &\tr U_{33}\\
 &  \mbox{s. t.} & U_{11}=1  \\
 &   & U_{22}+2U_{13}=0  \\
 &   & U_{33}+2U_{12}=0  \\
  && U \succeq 0.
    \end{array}
\]
Then the (unique) optimum pair is:\\
 \[U=\Diag(1,0,0), \quad
 x=[0,0,0], Z=\Diag(0,0,1)
\]
~~\\
~~\\
~~\\
But: $U+Z$ is not positive definite
\end{exam}
\end{slide}
\begin{slide}{}

\subsubsection{Overdetermined System of Optimality Conditions}
  \begin{eqnarray*}
    \A^*(y)+Z &=& C,\\
    \A(X) &=& b,\\
    ZX &=& \mu I.
  \end{eqnarray*}

unfortunate consequence:\\
 residual ($ZX-\mu I$) is not symmetric unless $X$ and $Z$ commute\\
therefore Newton step impossible

\end{slide}
\begin{slide}{}


due to success in linear programming codes,\\
 {\bf current philosophy - symmetrize?? But - correct???}

 E.g. AHO direction:  project onto subspace of symmetric matrices
\begin{displaymath}
  H(M) := \frac{1}{2}[M+M^T].
\end{displaymath}

\emph{Unscaled Symmetric System},
\[
\begin{array}{rclr}
    \A^*(dy)+dZ &=& -(\A^*(y_y)+Z_k-C) &=: -F_d,\\
    \A(dX) &=& -(\A(X_k)-b) &=: -f_p,\\
    H(Z_kdX + dZX_k) &=& -H(Z_kX_k-\mu I) &=: -H(F_c).
\end{array}
\]


\end{slide}
\begin{slide}{}


Monteiro-Zhang family of directions \cite{MontZhang:96}:\\
scale the problem, nonsingular matrix $P$
\beq\label{P_scaling}
   X := P X P^T,
  \quad  A_i := P^{-t}A_iP^{-1},
  \quad  C := P^{-t}CP^{-1}. 
\eeq
\emph{scaled primal problem}
\beq\label{eq:scaled_primal}
  \minpgms{\inner{ C}{ X}}{ \A( X) = b, 
     X\succeq 0},
\eeq

\emph{scaled dual problem} using 
$y := y,  Z := P^{-t}ZP^{-1}$
\beq
  \label{eq:scaled_dual}
  \maxpgms{\inner{b}{ y}}
  { \A^* (y)+ Z =  C,  Z\succeq 0}.
\eeq



\end{slide}
\begin{slide}{}


symmetric direction for family of parameterized programs
\[
\begin{array}{rclr}
     \A^*(dy)+ dZ &=& -( \A^*( y_k)+ Z_k- C) &=: - F_d,\\
     \A(dX) &=& -( \A( X_k)-b) &=: - f_p,\\
    H( Z_kdX + dZ X_k) &=& -
    H( Z_k X_k-\mu I) &=: -H( F_c),
\end{array}
\]
where
\[
\begin{array}{rclcl}
     F_d &=& 
     \A^*( y_k)+ Z_k- C &=& P^{-t}
    F_d P^{-1},\\
     f_p &=&  \A( X_k)-b &=& f_p,\\
     F_c &=&  Z_k X_k-\mu I &=& P^{-t} F_c P^{-1}.
\end{array}
\]


\end{slide}
\begin{slide}{}


\begin{table}[htbp]
% \begin{center}
    \leavevmode
    \begin{tabular}{|c|c|c|}\hline
      $P$ & Direction & Solvers\\
      \hline
      $I$ & AHO \cite{AlHaOv:94}& SDPPack,SDPA,SDPT3\\
      $Z^\half$& HKM
      \cite{HeReVaWo:93,KoShHa:94,Mont:95}&CSDP,SDPA,SDPT3\\
      $[X^\half(X^\half ZX^\half)^{-\half}X^\half]^\half$ & NT
      \cite{NesterovTodd:97} &SDPA,SDPT3,SeDuMi \\
      \hline
    \end{tabular}
    \caption{Instances of Monteiro-Zhang scaling matrix $P$ of equation
      (\ref{P_scaling}).}
    \label{tab:directions}
% \end{center}
\end{table}


\end{slide}
\begin{slide}{}


\begin{table}[htbp]
  \begin{center}
    \begin{tabular}{|c|l|}\hline
      CSDP &  {http://www.nmt.edu/\verb+~+borchers}\\
      SDPA & {http://www-neos.mcs.anl.gov/neos/solvers/SDP:SDPA}\\
      SDPPACK & {http://cs.nyu.edu/cs/faculty/overton/sdppack/sdppack.html}\\
      SDPT3 & http://www.math.cmu.edu/\verb+~+reha/sdpt3.html\\
      SeDuMi & {http://www2.unimaas.nl/\verb+~+sturm/research.html}\\ \hline
    \end{tabular}
    \caption{Major Semidefinite Solvers}
    \label{tab:implementations}
  \end{center}
\end{table}


\label{endsect:theory}





\end{slide}
\begin{slide}{}




\subsubsection{Historical Notes}
\label{sect:histnotes1}
$\bullet$
currently most active area in optimization (see 
HANDBOOK OF SEMIDEFINITE PROGRAMMING:
Theory, Algorithms, and Applications, 2000, \cite{SaVaWo:97}, for
comprehensive results, history, references, ...)

$\bullet$
Lyapunov
over 100 years ago on stability analysis of differential equations

$\bullet$
Bohnenblust 1948 on geometry of the cone of SDPs, \cite{Bohnen:48}

$\bullet$
Yakubovitch in the 1960's and Boyd
 and others on convex optimization in control in the 1980's,
e.g. solving Ricatti Equations
(called LMIs), e.g. \cite{BoB:91},\cite{VanBoy:94,VanBoy:94b}



\end{slide}
\begin{slide}{}


$\bullet$
matrix completion problems
(another name for SDP) started early 1980's,
continues to be a very active area of research,
\cite{DymGoh:81},\cite{GrJoSaWo:84},
surveys: \cite{joh:90,infieldsLaur:97}

$\bullet$
combinatorial optimization applications 1980's:
Lov\'{a}sz {\em theta function} \cite{Lo:79},
more recently the strong approximation results for the max-cut
problem by Goemans-Williamson, e.g. \cite{Goemans},
survey papers: \cite{Goem:97,Goemans98},\cite{Rendl:97}

$\bullet$
linear complementarity problems
can be extended to problems over the cone of semidefinite matrices,
e.g. \cite{MR96h:90117,MR1430560,KoShHa:94,KoShShi:95,MoTu:96}.


\end{slide}
\begin{slide}{}

$\bullet$
{Complexity, Distance to Ill-Posedness,
 and Condition Numbers}
SDP is a convex program and it can be solved 
to any desired accuracy in polynomial time, 
see seminal work of Nesterov and
Nemirovski e.g. 
\cite{MR89k:90130,int:Nesterov4,int:Nesterov5,NeN:90,NeN:88,NeN:90b,NeN:91}.
Another measure of complexity is the distance to ill-posedness:
 e.g.  work by Renegar
\cite{Renegar:92,MR97g:90134,MR96i:90029,MR96c:90048,MR95d:65116}.



$\bullet$
{Cone Programming}
this is a generalization of SDP,
also called {\em generalized linear programming},
in paper by Bellman and Fan 1963, \cite{FanBell:63}. 
Other books deal with problems over cones date back to 60s, e.g.
\cite{MR55:13315},\cite{Hol:75,Lu:69,Jame:70,Ja:86,MR37:3315}.
More recently, generalization of SDP to 
more general cones,
e.g.  G{\" u}ler and Tuncel, \cite{GulerTuncel:98,TunXu:99} and also
Hauser \cite{Hauser:99}



\end{slide}
\begin{slide}{}

$\bullet$
Other Related Areas e.g.:
Eigenvalue Functions, e.g. \cite{CuDoWo:75}, \cite{Ov:88,Ov:92};
{Financial Applications};
{Generalized Convexity}, e.g.  \cite{MR83g:90075,MarOlk:79};
Statistics; Nonlinear Programming; ...


\label{endsect:basicpropnot}
\end{slide}
\begin{slide}{}


\subsection{Motivation/Examples/Applications}
\label{sect:motexampappl}

\begin{enumerate}

\item[$\bullet$]
Quadratic constrained quadratic programs
\item[$\bullet$]
Lov\'{a}sz theta function
\item[$\bullet$]
Statistics
\item[$\bullet$]
minimizing the $L_2$-operator norm of a matrix
\item[$\bullet$]
linear programming
\item[$\bullet$]
robust mathematical programming
\item[$\bullet$]
engineering, e.g. control theory
\item[$\bullet$]
Combinatorial Problems, e.g. the Max-Cut Problem

\end{enumerate}

\end{slide}
\begin{slide}{}
\subsubsection{Quadratic Constrained Quadratic Programs}
\label{sect:qqpintro}

{What is SEMIDEFINITE PROGRAMMING?\\ Why use it?}

$\bullet$ 
Quadratic approximations are better than linear approximations. \\
(For example, model $x \in \{0,1\}$ using $x^2-x=0$.)\\
And, we
can solve relaxations of
quadratic approximations efficiently using semidefinite
programming.
\end{slide}
\begin{slide}{}


\subsubsection{Quadratic Matrix Inequalities}
\label{sect:quadmatrix}
Let $X$ be $k \times l$ and
$F(X)=(AXB)AXB)^T+CXD+(CXD)^T+E$. Then
\[
\begin{array}{l}
\left\{ (X,Y) : F(X) \preceq Y \right\}=\\
\qquad
\left\{ (X,Y) : 
\pmatrix{ I  & (AXB)^T \cr (AXB) &  \left( Y - E -CXD- (CXD)^T\right)}
\succeq 0
\right\} 
\end{array}
\]

(by Schur Complement Lemma)

\end{slide}
\begin{slide}{}

\subsubsection{Nonnegative Polynomial Approximation}
\label{sect:poynapprox}
 
(e.g. \cite{nesterov97,GenHachNestvandooren:00})
best polynomial $p(t)$ approximation to a given function $f(t)$, where
$p(t) \in \pols^+_{2k}(\RR)$,
i.e. nonnegative, degree $\leq 2k$

$p(t)=\sum\limits_{i=0}^k p_it^i \equiv (p_0, p_1, \ldots p_{2k})^T$

Then
\[
p \in \pols^+_{2k}(\RR) \quad \underline{\mbox{  iff  }}\quad
p(t)=e(t)^TXe(t), \mbox{ for some } X \succeq 0,
\]
where $e(t)=\left(1,t,t^2, \ldots, t^k \right)$





\end{slide}
\begin{slide}{}


\subsubsection{Lov\'{a}sz theta function, \cite{Lo:79}}
\label{sect:lovtheta}
Let $G=(V,E)$ be a {\bf \em simple undirected graph}\\
$~~~~~~~~~~~~~~~~~~$(node set $V$, edge set $E$)

a {\bf \em clique} - is subgraph of $G$, $\exists$ edge between every pair of nodes

{\bf \em  number of colours} to colour clique $\geq$ size of the clique 


{\bf \em $\chi (G)$ - chromatic number},
least number of colours for $G$

{\bf \em size of maximum clique} $\omega (G)$  
\[\omega (G)  \leq \chi (G) \quad \mbox{two {\bf hard} to compute integers}\]

{\bf \em strict inequality}, for example: $G$ simple circle with $\|V\|=5$, 
then: $2=\omega(G) < \chi(G)=3$.  


\end{slide}
\begin{slide}{}

Let $A=A(G)$ be {\bf \em adjacency matrix} of $G$:\\
 i.e.  $A_{ij}=1$ if $(i,j)\in E$ and 0 otherwise.

The largest eigenvalue of  $A$, 
$\lambda_{\max}(A) = \max\limits_{||x||=1} x^TAx$
 is a convex function of its elements, $A_{ij}$.
Fix $A_{ij}=1$ if $(i,j)\in E$, otherwise leave the elements free,
i.e. define the \emph{Lov\'{a}sz theta number} 
\[\vartheta(G) := \min \left\{ \lambda_{\max}(X) : X=X^T,
                       X_{ij}=1 \mbox{ if } (i,j)\in E \right\}
\]


\end{slide}
\begin{slide}{}


\begin{lem}
\[ \mbox{size of max clique} \quad \omega(G)\leq\vartheta(G) 
 \quad \mbox{Lov\'{a}sz theta number}
\]
\end{lem}

\bpr
Rename vertices of $G$, such that ${1,2,\ldots,k}$ is the maximum clique of $G$.
This is a permutation of the rows/columns of the adjacency matrix $X$, 
so the eigenvalues of $X$ are unchanged. We get
\[ X=\left( \begin{array} {cc}
J_k & X_2 \\
X_3 & X_4 \end{array} \right ), \qquad
J_k= E \in \mathbb{R}^{k \times k}.
\]
Therefore
\[\lambda_{\max}(J_k)=\tr (J_K)=\tr (e e^T) =\tr (e^T \cdot e) =k\]
Now use interlacing of eigenvalues
\[ \lambda_{max}(J_k) \leq \lambda_{max}(X)\].
\epr
\noindent
In addition, it can be shown that 
\[
\mbox{size of max clique} \quad \omega(G) \leq \vartheta(G) \leq \chi(G)
\quad \mbox{size of chromatic number} 
\]
  $\vartheta (G)$ gives a lower bound of $\chi(G)$ 
and an upper bound of $\omega(G)$. \\
(bounds for two NP-hard numbers)

And, as shown in Section~\ref{sect:prelexamples}, we can use SDP to solve
the min-max eigenvalue problem (efficiently).

\end{slide}
\begin{slide}{}


\subsubsection{Generalized Eigenvalue Problems for $X=X^T$}
\label{sect:geneigprobs}
\begin{enumerate}
\item {\bf The Generalized Eigenvalue Problem}

Let $M,A$ be two $n \times n$ symmetric matrices, $M \succ 0$.

The set of eigenvalues of the {\em matrix pencil}, denoted $[M,A]$, is
$\{ \lambda \in \RR : \lambda M - A \mbox{ is singular} \}$.
\[ \lambda_{\max}([M,X]) \leq t \quad \mbox{is equivalent to}\quad
tM-X \succeq 0.
\]

\item
{\bf Spectral Norm of Symmetric $X$}
\[
\left\{ | \lambda_i(X) | \leq t, ~ \forall i  \right\}
\mbox{ iff }
\left\{  tI - X \succeq 0, ~ tI + X \succeq 0   \right\}
\]
\item
{\bf Sum of $k$ largest eigenvalues of Symmetric $X$}\\
Let $S_k(X)$ denote the sum of the largest $k$ eigenvalues of $X$.
\[
S_k(X) \leq t \underline{\bf \mbox{ iff }}
\begin{array}{rcl}
t-ks-\tr Z \succeq 0, Z \succeq 0, Z-X+sI \succeq 0.
\end{array}
\]

\end{enumerate}


\end{slide}
\begin{slide}{}


\subsubsection{SDP Application in Statistics}
\label{sect:stats}
If $m_0, m_1, m_2, \ldots , m_{2n}$ are moments of some distribution, then
\[ H(m_0, m_1, \ldots , m_{2n})=\left[ \begin{array}{cccc}
m_0 & m_1 & \cdots & m_n \\
m_1 & \cdots & \cdots & m_{n+1} \\
\cdots & \cdots & \cdots & \cdots \\
m_n & m_{n+1} & \cdots & m_{2n} \end{array} \right ] \succeq 0 \]
Find distribution with maximal variance and $l_i\leq m_i \leq u_i$:
\begin{eqnarray*}
\max & y \\
{\mbox{s.t.}} & \left[ \begin{array}{cc}
m_2-y & m_1 \\
m_1 & 1 \end{array} \right] \succeq 0 \\
 & l_i \leq m_i \leq u_i  \, (i=1, \ldots 2n) \end{eqnarray*}


\end{slide}
\begin{slide}{}

\subsubsection{Minimizing the $L_2$-operator Norm of a Matrix}
\label{sect:opernorm}
\subsubsection{Linear Programming}
\label{sect:LP}
\subsubsection{Robust Mathematical Programming}
\label{sect:robust}




\end{slide}
\begin{slide}{}


\subsubsection{Control Theory}
\label{sect:cntrl}


\end{slide}
\begin{slide}{}


\subsubsection{Max-Cut Problem}
\label{sect:combprobs}

(MC) is a combinatorial optimization problem on
undirected graphs with weights on the edges.
\begin{prob}
Find a partition of the set of vertices
into two parts that maximizes the sum of the
weights on the edges that have one end in each part of the partition.
\end{prob}



\end{slide}
\begin{slide}{}

\begin{center}
{\bf Quadratic Model of MC}
\end{center}

given graph $G$ with vertex set $\{1,\dots,n\}$\\
weighted adjacency matrix $A(G)=(a_{ij})$, weight on edge $ij$\\
{\em Laplacian matrix}
$L := \Diag(A(G) \cdot e) - A(G)$\\
Let $v \in \{\pm 1\}^n$ represent any cut in the graph
\[\label{mc}
(\mbox{MC}) \quad
        \begin{array}{ccl}
        \mu^* =&\max   & \frac{1}{4} v^{T} Lv \\
        &\mbox{s.t.} & v_i^2 = 1,\quad  i=1,\ldots,n,
        \end{array}
\]
where $\mu^*$ denotes the optimal value of MC.



\end{slide}
\begin{slide}{}

\begin{center} {\bf Some applications of MC} \end{center}

\vspace{10mm}

\begin{itemize}
\item Statistical physics: Finding the ground state of a spin glass
according to the Ising model
\item VLSI: Minimizing the number of vias in a two-sided circuit board
\item Network design: Solving the separation problem in a cutting plane
approach
\end{itemize}
\end{slide}

\begin{slide}{}
\begin{center} {\bf Spin glasses} \end{center}

A {\em spin glass} is an alloy of magnetic impurities diluted in a
nonmagnetic metal.

Suppose we have a spin glass subjected to an exterior magnetic field.
At a very low temperature (close to $0^{\mbox{{\small o}}}$K)
the spin glass
attains a minimum energy configuration, called the ground state.

This state can be found by {\em minimizing} the hamiltonian representing
the total energy of the system.
\end{slide}

\begin{slide}{}
\begin{center} {\bf Ising model} \end{center}

Suppose the spin glass has $n$ impurities, or atoms.

Let $v_i \in \{\pm 1\}$ denote the magnetic orientation of atom $i$
(``north/south''), and $v_0$ denote the magnetic orientation of the 
exterior magnetic field which has strength $h$.

For any pair $i,j$ of atoms, $J_{ij}$ denotes the interaction
between them, e.g.
\[
J_{ij} = A \frac{\cos(D r_{ij})}{B^3 r_{ij}^3},
\]
where $r_{ij}$ is the distance between the atoms, and $A,B,D$ are
given constants.
\end{slide}

\begin{slide}{}
Then the energy of the system is modelled by the hamiltonian

\[
H := - \sum\limits_{1 \leq i < j \leq n} J_{ij} v_i v_j 
     - h \sum\limits_{j=1}^{n} v_0 v_j.
\]

To find the ground state, we want to minimize $H$ over the $2^n$ possible
configurations of the Ising spins (assuming $v_0 = +1$ wlog).

Equivalently, we can maximize $-H$.
\end{slide}

\begin{slide}{}
\begin{center} {\bf MC Formulation} \end{center}

Let $G=(V,E)$ be a complete graph with $V=\{0,1,\ldots,n\}$ and

$
\begin{array}{l}
w_{ij} = -J_{ij}, \,\mbox{for}\; 1 \leq i < j \leq n, \\
w_{0j} = -h, \,\mbox{for}\; 1 \leq j \leq n.
\end{array}
$

Then

$
\begin{array}{ccl}
H  & = & \sum\limits_{0 \leq i < j \leq n} w_{ij} v_i v_j \\
&=& \sum\limits_{0 \leq i < j \leq n : v_i=v_j} w_{ij} v_i v_j
  + \sum\limits_{0 \leq i < j \leq n : v_i \not= v_j} w_{ij} v_i v_j \\
&=& \sum\limits_{ij \not\in \delta(S)} w_{ij}
  ~~ - \sum\limits_{ij \in \delta(S)} w_{ij}, \\
\end{array}
$

where $S = \{ i \in V : v_i = +1 \}$
and $\delta(S) = \{ ij \in E : i \in S, j \in V \backslash S \}$.
\end{slide}

\begin{slide}{}
Now define the constant $K := \sum\limits_{ij \in E} w_{ij}$.

Then
\[
H-K = - 2 \sum\limits_{ij \in \delta(S)} w_{ij}
\]

and so $-H = 2 \sum\limits_{ij \in \delta(S)} w_{ij} - K$.
So our equivalent optimization problem is:

$
\begin{array}{cl}
\mbox{max} & \sum\limits_{ij \in \delta(S)} w_{ij} \\
\mbox{s.t.}& S \subseteq V\\
\end{array}
$

which is a MC problem.

\end{slide}

\begin{slide}{}
\begin{center} {\bf Approaches to MC} \end{center}

The general MC problem is NP-hard,
even though it is tractable for some classes of graphs, e.g.
planar graphs.
In fact, solving within relative error of 1 percent is NP-hard.

Techniques used for general MC include:
\begin{itemize}
\item Heuristics;
\item Integer programming enumerative techniques (Branch-and-Bound);
\item Approximation algorithms.
\end{itemize}


We seek {\em SDP relaxations} of MC (solvable in polynomial-time)

that yield tight (upper) bounds on the optimal value of MC.



\end{slide}

\begin{slide}{}

\begin{center} {\bf One Formulation of MC} \end{center}

Let $v \in \{\pm 1\}^n$, $n = | V |$,
represent a cut in the graph $G$
via $S = \{i : v_i = +1\}$ and $V \backslash S = \{i : v_i = -1\}$.

Then we can formulate MC as:

$\begin{array}{cll}
\mu^{*}=&\max
    &\sum\limits_{1\leq i<j \leq n } w_{ij} \left(\frac{1-v_i
v_j}{2}\right) \\
                    &\mbox{s.t.}
    & v \in \{ \pm 1\}^n.
\end{array}$

Equivalently,
\[
(\mbox{MC1}) \quad
\begin{array}{cll}
\mu^{*}=
   &\max & v^{T} Q \, v \\
   &\mbox{s.t.}  & v_i^2 = 1, \hspace{6mm} i=1,\ldots,n,
        \end{array}
\]

where $Q = \frac{1}{4} (\Diag(A\,e)-A)$ (Laplacian $L$), 
and $A=(w_{ij})$ is the weighted adjacency matrix of $G$.


\end{slide}

\begin{slide}{}

With $Q := \frac{1}{4} L$, $X := vv^{T}, v \in \{\pm 1\}^n$, \\
then $v^T Q v=\tr QX$ and equivalent formulation is:
\[\label{mcmatrform}
(\mbox{MC1}) \quad
        \begin{array}{ccl}
        \mu^*=&\max        & \tr QX \\
                    &\mbox{s.t.} & \diag(X)= e\\
                                && \mbox{rank}(X)=1\\
                                && X\succeq 0, X \in \Sn,
        \end{array}
\]

{\em relax} by deleting the {\em hard} constraint $\mbox{rank}(X)=1$
(get {\em elliptope})

\end{slide}
\begin{slide}{}

\begin{figure}[h]
\centering\psfig{file=laurpol.ps,height=70mm}
\end{figure}
{\bf Elliptope for n=3, \cite{LaPo:94} }

\end{slide}
\begin{slide}{}


Goemans \& Williamson proved that

if $w_{ij} \geq 0\,\,\forall\: i,j$, then
\[
\mu^* \geq \alpha \, \nu_1^*,
\]
where $\alpha = \min_{0 \leq \theta \leq \pi} \frac{2}{\pi}
 \frac{\theta}{1-\cos \theta}
 \approx 0.87856$.

\vspace{1cm}

Since $\frac{1}{\alpha} \approx 1.13823 \leq 1.14$, this implies
\[
\mu^{*} \leq \nu_1^* \leq 1.14\:\mu^*.
\]

\end{slide}

\begin{slide}{}
Other such convex relaxations have been studied before.

\vspace{1cm}

The smallest convex set containing all the
rank-one matrices $X$ corresponding to cuts is the {\it cut polytope}:
\[
C_n := \mbox{Conv}\{X : X=v v^T, v \in \{\pm1\}^n\}.
\]

In fact,
\[
\begin{array}{cll}
\mu^{*}=
    &\max & \trace Q X \\
    &\mbox{s.t.} & X \in C_n. \\
\end{array}
\]

However, it is not known how to optimize in polynomial-time over $C_n$.


\end{slide}
\begin{slide}{}

\begin{figure}[h]
\centering\psfig{file=cutc3.ps,height=70mm}
\end{figure}
{\bf Cut polytope for n=3}


\end{slide}
\begin{slide}{}

Another convex relaxation is the
{\it metric polytope} $M_n$, defined by

$
\begin{array}{l}
M_n := \{X \in \Sn : \diag(X) = e,\:\mbox{and}\: \\
~~~   X_{ij}+X_{ik}+X_{jk} \geq -1, X_{ij}-X_{ik}-X_{jk} \geq -1, \\
~~  -X_{ij}+X_{ik}-X_{jk} \geq -1,-X_{ij}-X_{ik}+X_{jk} \geq -1, \\
~~~    \forall\, 1 \leq i < j < k \leq n \}.
\end{array}
$

These inequalities model the fact that for any three mutually
connected vertices in the graph, it is only possible to cut
either zero or two
of the edges joining them.

It is known that:

$C_n = M_n$ for $n \leq 4$,
but $C_n \varsubsetneq M_n$ for $n \geq 5$.


\label{endsect:motexampappl}


\label{endsect:intromotiv}


\end{slide}
\begin{slide}{}


\bs{Algorithms}
\label{sect:algor}
\subsection{Outline}
\begin{enumerate}

\item[$\bullet$]
Primal-dual framework
\item[$\bullet$]
Gauss-Newton methods
\item[$\bullet$]
Exploiting Sparsity

\end{enumerate}


\end{slide}
\begin{slide}{}


\subsection{Primal-Dual Framework}
\label{sect:pdframewk}

  {\bf Generic Interior-Point Algorithm for Monteiro-Zhang family}
\begin{enumerate}
  \item[]
  \begin{enumerate}
    \item[$\bullet$] Given $\epsilon>0$ \hfill{\em Tolerance}
    \item[$\bullet$] Given $X,y,Z$ \hfill{\em Must satisfy some condition}
    \item[$\bullet$] $\mu=\frac{\inner{Z}{X}}{n}$ \hfill{\em Initial barrier parameter}
    \item[$\bullet$]
    {\bf WHILE}~~{$\mu > \epsilon$}
  \begin{enumerate}
    \item[$\bullet\bullet$] Choose scaling $P$\hfill{\em Possibly dependent on $X,Z$}
    \item[$\bullet\bullet$] Choose centrality $0<\tau<1$ \hfill{\em According to some condition}
    \item[$\bullet\bullet$] $\mu \leftarrow \tau \frac{\inner{Z}{X}}{n}$ \hfill{\em Update target}
    \item[$\bullet\bullet$] Solve ({\em scaled ahosystem})\hfill {\em Scaled AHO direction}
    \item[$\bullet\bullet$] Choose steplength $\alpha$\hfill{\em To maintain
    positive definiteness}
    \item[$\bullet\bullet$] $X=X+dX, y=y+dy, Z=Z+dZ$\hfill{\em Update iterate}
  \end{enumerate}
    \item[$\bullet$]{\bf ENDWHILE}
  \end{enumerate}
\end{enumerate}



\end{slide}
\begin{slide}{}
\subsubsection{Common LP,SDP Framework}
\label{sect:commonframewk}

For  LPs, Lor\'{e}ntz cone problems, SDPs:
use Newton's method to get step at each iteration.  

Common system of equations at each step:
\beq\label{System for all}
\left( \begin{array}{ccc}
\mathcal{A} & 0 & 0 \\
0 & \mathcal{A}^* & I \\
\mathcal{E} & 0 & \mathcal{F} 
\end{array} \right)
\left(  \begin{array}{c}
\Delta X \\
\Delta y \\
\Delta Z 
\end{array} \right)
=
\left( \begin{array}{c}
\F_p\\
\F_d\\
\F_c
\end{array} \right)
\eeq



$\Delta x$, $\Delta y$, $\Delta Z$ step; $\F_p$, $\F_d$, and $\F_c$ residuals 
primal, dual, complementarity



\end{slide}
\begin{slide}{}


Assume $\mathcal{E}$ and $\mathcal{AE}^{-1}\mathcal{FA}^*$ are invertible. 
Then the solution to (\ref{System for all}) is:
\beq\label{Solution to all}
\begin{array}{lll}
\Delta y & = & (\mathcal{AE}^{-1}\mathcal{FA}^T)^{-1}(\mathbf{F_p}-
        \mathcal{AE}^{-1}(\mathbf{F_c}-\mathcal{F}\mathbf{F_d})) \\
\Delta Z & = & \mathbf{F_d}-\mathcal{A}^*\Delta y \\
\Delta X & = & \mathcal{E}^{-1}(\mathbf{F_c}-\mathcal{F}(\mathbf{F_d}
    -\mathcal{A}^*\Delta y))
\end{array}
\eeq


\end{slide}
\begin{slide}{}

{\bf LP Case}

central path satisfies
\beq\label{LPxs}
\begin{array}{rcl}
\A X & = & b \\
\A^* y+ Z  & = & C \\ 
X_i Z_i  & = &  \mu  ~ i=1, \ldots, n,
\end{array}
\eeq
where $X=x, Z=z$.
Newton's method on \req{LPxs} yields \req{System for all}, 
where $\mathcal{A}=A$, $\mathcal{E}=\Diag({z})$, $\mathcal{F}=\Diag({x})$, 
$\mathbf{F_p}=b-A x$, $\mathbf{F_d}=c-A^T y- z$, 
$\mathbf{F_c}_i=\mu-x_i z_i,~ i=1, \dots,n$.

\begin{remark}
The main work is the computation of 
(the positive definite Schur complement) 
$(\mathcal{AE}^{-1}\mathcal{FA}^T)^{-1}$.  
\end{remark}



\end{slide}
\begin{slide}{}


{\bf SDP Case}

Use $(-\ln \det X)$ for barrier function.
Then its gradient is $-X^{-1}$, its Hessian is $X^{-1}\otimes X^{-1}$.  
(So convex function on $\p$.) The log-barrier problem is:
\begin{eqnarray*}
\min_X & \trace C X-\mu \ln \det X \\
\mbox{s.t.}& \trace A_i X=b_i \, ~i=1, \cdots , m
\end{eqnarray*}



\end{slide}
\begin{slide}{}

Let $y$ be the Lagrange multiplier; Lagrangian is
\[\mathcal{L}(X,y)= \trace C X-\mu \ln \det X +  y^T(b - \A X ).
\].  

KKT system:
\begin{eqnarray*}
\nabla_X \mathcal{L} & = & C-\mu X^{-1}-\A^* y=0 \\
\nabla_{y} \mathcal{L} & = & b-\A X =0,~ i=1, \dots,m
\end{eqnarray*} 



\end{slide}
\begin{slide}{}


Define $Z:= \mu X^{-1}$. 
The system of equations on the central path is now:
\beq
\begin{array}{rcl}
A_i \bullet X & = & b_i \; (i=1, \cdots m) \\
\sum_{i=1}^m y_iA_i +S & = & C \\
\mbox{(one of)} & &
\left\{
\begin{array}{rcl}
S & =   \mu X^{-1} \\
X & =  \mu S^{-1} \\
XS & = \mu I
\end{array}
\right.
\end{array}
\label{eq:SDP}
\eeq



\end{slide}
\begin{slide}{}

Use $XZ=\mu I$ and apply Newton's method to the system:
\beq
\begin{array}{rclc}\label{eq:SDPXZ} 
\A \Delta X  &=&  b- \A X 
     & (=(\mathbf{F_p})_i)   \\
\A \Delta y +\Delta Z  &=&  C-S -\A^*y 
         & (=F_d)  \\


X \Delta Z + \Delta X Z  &=&  \mu I -XZ  & (=F_c) 
\end{array}
\eeq



To write \req{eq:SDPXZ} into the common form of \req{System for all}, 
one should first turn matrix into vector.  
For a $p\times q$ matrix $B$, define $\kvec B$ to be a $pq$ column vector 
made up of columns of $B$ stacked on each other.  
Another way is to use $\svec$.  
Since here the matrices $X$ and $Z$ are symmetric, 
only the lower triangle of the matrices need to be saved.  
Let $\svec$ denote such an operator that maps $\mathbb{R}^{n\times n}$ 
to $\mathbb{R}^{\frac{n(n+1)}{2}}$.  
For the reason of simplicity, we use $kvec$ for now.


\end{slide}
\begin{slide}{}


\begin{definition}
\underline{Kronecker product($\otimes$):} 
If $A\in \mathbb{R}^{m\times n}$, $B \in \mathbb{R}^{p\times q}$, 
then $A \otimes B \in \mathbb{R}^{mp \times nq} $,
\begin{displaymath}
A\otimes B :=
\pmatrix{
a_{11}B & a_{12}B & \cdots & a_{1n}B \cr
\vdots & \vdots & \vdots & \vdots \cr
a_{m1}B & a_{m2}B & \cdots & a_{mn}B}
\end{displaymath}
\end{definition}

\begin{exam}
If
\[ A=
\pmatrix{
1 & 2 \cr
3 & 4
}
\mbox{ and }
B=
\pmatrix{
5 & 6
}
\]
then
\[
A\otimes B =
\pmatrix{
5 & 6 & 10 & 12 \cr
15 & 18 & 20 & 24}
\]
\end{exam}



\end{slide}
\begin{slide}{}


Fundamental properties of Kronocker product:
\begin{enumerate}
\item
\[
\kvec(ABC)=(C^T\otimes A)\kvec(B) \mbox{ and } 
      (A\otimes B)(C \otimes D)=(AC)\otimes (BD)
\]
\item
 $(A\otimes B)^T=A^T \otimes B^T $
\item
$ (A\otimes B)^{-1}=A^{-1}\otimes B^{-1} $
\item
$ \Diag(x) \otimes  \Diag(y)  = \Diag( x \otimes y)$
\item
Suppose $A=P^{-1}(\Diag(\lambda_i))P$,  and 
      $B=Q^{-1}(Diag(\omega_j))Q$, then
 $A \otimes B =(P^{-1}\otimes Q^{-1})(\Diag(\lambda_i \omega_j))(P \otimes Q) $
\item
Let $\lambda_i$ be the $i$-th eigenvalue of  $A$, and
$\omega_j$ be the $j$-th eigenvalue of $B$, then 
the $ij$-th eigenvalue of  $A\otimes I + I \otimes B$
is $\lambda_i+\omega_j$
\end{enumerate}



\end{slide}
\begin{slide}{}


$X\Delta Z +\Delta X Z =\mu I -XZ$,
can be rewritten using  Kronocker products:
\[
(Z\otimes I)\kvec \Delta X + (I \otimes X)\kvec \Delta Z 
   =\kvec (\mu I -XZ) =\kvec(F_c)
\]


%Let $ith$ row of $\A$ be $(vec A)^T$, $\E=S\otimes I$, $\F=I\otimes X$, $\mathbf{r_P}=b-\A vec(X)$, $vec(R_d)=vec(C-S-mat(\A^T \vy))$, \req{eq:SDPXSp}---\req{eq:SDPXSc} now can be written as:
%\begin{displaymath}
%\begin{pmatrix}
%\A & 0 & 0 \\
%0 & \A^T & I \\
%\E & 0 & \F
%\end{pmatrix}
%\begin{pmatrix}
%vec(\Delta X) \\
%\dy \\
%vec(\Delta S)
%\end{pmatrix}
%=
%\begin{pmatrix}
%\mathbf{r_p} \\
%vec(R_d) \\
%vec(R_c)
%\end{pmatrix}
%\end{displaymath}
%Notice that $\E$ are linear in $S$, and $\F$ are linear in $X$, $M=\A S^{-1} \otimes X \A^T$.
%>From \req{eq:SDPXSd}, one can see that $\Delta S$ is symmetric, but $\Delta X$ is not necessarily symmetric.  (\req{eq:SDPXSc} can be written as $\Delta X S=\mu I -X(S+\Delta S)$, for two symmetric matrices $C$ and $D$, $CD$ is not necessarily symmetric).\\ 
%\textbf{$XS$ Method:} One way to solve this problem is at each iteration, replace the iterate $\Delta X$ with $\frac{\Delta X + \Delta X^T}{2}$, we call it $XS$ method. \\ 
%\textbf{$SX$ Method:} One can replace $XS=\mu I$ with $SX=\mu I$, and symmetrize each $X$ iterate.  This method differs from $XS$ method only on $R_c$, but technically, they are different directions. \\
%In both $XS$ method and $SX$ method, $M$ is symmetric and positvie definite in the interior(Assume $\A$ has full rank), so Cholesky factorization can be applied to them.  The drawback of these methods is that they deviate from Newton direction, since $X$ iterate is changed.  Another way is to use $S=\mu X^{-1}$ or $X=\mu S^{-1}$ in \req{eq:SDP}. \\ 
%\textbf{$X$ Method:} use $S=\mu X^{-1}$, then $\E=\mu X^{-1} \otimes X^{-1}$, $\F=I$ \\
%\textbf{$S$ Method:} use $X=\mu S^{-1}$, then $\E=I$, $\F=\mu S^{-1} \otimes S^{-1}$ \\
%$X$ method and $S$ method are pure Newton's methods, and $M$ in both cases are symmetric.  But both methods are numerically unstable near boundary as they involve $X^{-1}$ or $S^{-1}$. Besides, the systems of equations are not symmetric in both cases.\\
%


{\bf $XZ+ZX$ Method to guarantee symmetry (AHO Direction)}
\[ \mbox{replace } XZ=\mu I \mbox{ with } \frac{XZ+ZX}{2}=\mu I.
\]

To justify this method, one can show that for
$X \succeq 0,  Z \succeq 0$,
\[
XZ=\mu I \Longleftrightarrow \frac{XZ+ZX}{2}=\mu I.
\]
%\begin{proof}
%$\Longrightarrow$ is trival. \\
%$\Longleftarrow$  : let $A= XS$, then $ \frac{XS+SX}{2}=\mu I \Longleftrightarrow \frac{A+A^T}{2}=\mu I$, which means
%\begin{displaymath}
%a_{ij}=
%\begin{cases}
%-a_{ji} & (i\neq j) \\
%\mu & (i=j)
%\end{cases}
%\end{displaymath}
%So $XS-\mu I$ is a skew symmetric matrix, which means all its nonzero eigenvalues are pure imaginaries( If $B=-B^T$, then $B^2=-BB^T$.  Since $BB^T$ is symmetric and positive semidefinite, all the eigenvalues of $B^2$ are negative.)
%At the same time, $XS-\mu I \approx \Xsq S \Xsq -\mu I$, which is symmetric, so all its eigenvalues are real.
%Thus, all the eigenvalues of $XS-\mu I$ are $0$.
%\end{proof}
%Apply Newton to $\frac{XS+SX}{2}=\mu I$, one can get:
%\[ X\Delta S +\Delta S X + S \Delta X + \Delta X S =2 \mu I -XS -SX \] 
%Since $vec(\Delta X S + S \Delta X + \Delta S X + X \Delta S)=(S \otimes I +I \otimes S)\Delta X +(X \otimes I + I \otimes X ) \Delta S $, for $XS+SX$ method, $\E=S\otimes I +I\otimes S$, and $\F=X\otimes I + I \otimes X$, $R_c=2\mu I -XS -SX$.
%Assume primal and dual nondegeneracy and strict complementary, then \req{System for all} is well defined for $XS+SX$ method.  And it is numerically more stable even near the optimum.  But $\E^{-1}\F$ is not symmetric any more, so instead of Cholesky factorization, LU factorization need to be employed.
%
%
%
%

\end{slide}
\begin{slide}{}



\subsection{Gauss-Newton Method}
\label{sect:GNmethod}
\subsubsection{Outline}
\begin{enumerate}

\item[$\bullet$]
Background for GN
\item[$\bullet$]
Properties of the Gauss-Newton Direction
\item[$\bullet$]
{Conditioning of the Jacobian}
\item[$\bullet$]
{Invariance}
\item[$\bullet$]
{Convergence of Gauss-Newton Method}
\item[$\bullet$]
Numerical Tests

\end{enumerate}



\end{slide}
\begin{slide}{}

\subsubsection{Background for GN Methods}
$H_1$, $H_2$ {\em TWO} real Hilbert spaces

$F:H_1 \to H_2$ a nonlinear operator 

solve the equation:
\beq \label{eq1} F(x)=0.
\eeq

{\bf Assumption}:\ 
Problem (\ref{eq1}) has a solution $\hat{x}$

 Newton's method (\cite{OrRhe:70}):
\beq \label{nm}
x_{k+1}=x_k-F^{\prime}(x_k)^{-1}F(x_k),
\eeq
where $F^{\prime}(x) : H_1 \to H_2$ -
Fr\'echet derivative of $F$ (linear operator) 

{\bf needs:} bounded invertibility of $F^{\prime}(x_k)$ \\

(existence of a bounded 
linear operator $[F^{\prime}(x)]^{-1}$ near $\hat{x}$)


\end{slide}
\begin{slide}{}

to avoid the restrictions use Gauss-Newton procedure
(\cite{OrRhe:70}):
\beq\label{eq2}
\begin{array}{rcl}
x_{k+1}&=& x_k-[F^{\prime *}(x_k)]^{\dagger} F(x_k)\\
&=&x_k-[F^{\prime *}(x_k)F^{\prime}(x_k)]^{-1}F^{\prime *}(x_k)F(x_k), 
\end{array}
\eeq
where $\cdot^{\dagger}$ denotes Moore-Penrose generalized inverse




\end{slide}
\begin{slide}{}
\subsubsection{Gauss-Newton Methods for SDP}
In numerical analysis, how do we
solve a nonlinear system of equations?

The standard approach is to solve the equivalent
 nonlinear least squares problem using
``Gauss-Newton'' method. 
\[  \min \frac 12 || F_{\mu} (X,y,Z) ||^2_2  \]
subject to $X,Z>0.$

\end{slide}
\begin{slide}{}
In the LP case,
when we differentiate we get a square system and Gauss-Newton
reduces to Newton's method.

i.e. solving
\[
\left( F_{\mu}^{\prime}\right)^T F_{\mu}^{\prime} (\Delta s)
   = - \left( F_{\mu}^{\prime}\right)^T F_{\mu}
\]
(where $F_{\mu}^{\prime}$ is the Jacobian)\\
 is equivalent to solving
\[
 F_{\mu}^{\prime} (\Delta s)
   = -  F_{\mu}
\]



\end{slide}
\begin{slide}{}
{\bf framework} p-d i-p:\\
{\bf Given} $(X^0,y^0,Z^0) \in {\cal F}^0$\\
{\bf for} $k=0,1,2 \ldots $\\
\hspace{.5in}{\bf solve} for the search direction\\
     \[
  ~~~~F_{\mu}^{\prime}(X^k,y^k,Z^k) 
\left( 
\begin{array}{ccc}
\Delta X^k \\ \Delta y^k \\ \Delta Z^k 
\end{array}
\right)
=  
\left( 
\begin{array}{ccc}
0 \\ 0 \\ -X^kZ^k + \sigma_k \mu_k I 
\end{array}
\right)
\]
\hspace{.5in}where $\sigma_k$ centering, $\mu_k=\tr X^kZ^k/n$

\[ \begin{array}{c}
(X^{k+1},y^{k+1},Z^{k+1}) =~~~~~~~~~~~~~~~~~~\\
~~~~~~~~~~~~~~~~ (X^k,y^k,Z^k) + \alpha_k
(\Delta X^k,  \Delta y^k,  \Delta Z^k)
\end{array}
\]
\hspace{.5in}so that $(X^{k+1},Z^{k+1}) \succ 0$\\
{\bf end (for)}.

\end{slide}
\begin{slide}{}
\begin{center}
Shrinking the Overdetermined System
\end{center}


We can find redundant equations and solve a
smaller system.  We can remove
the lower triangular part from the nonlinear equation and get
the equivalent equation
\[
\T \left(ZX  - \mu I \right)= 0,
\]
where the linear operator $\T: \Mn \rightarrow \Sn$
by ignoring the strictly lower triangular part of the matrix, i.e. the
$i,j$ components are
\[
\left(\T(W)\right)_{ij} = \left\{
\begin{array}{cc}
W_{ij} & \mbox{if}~ i \leq j \\
W_{ji} & \mbox{if}~ i > j \\
\end{array}
\right.
\]


The resulting system is now square.

$\T$ is an orthogonal projection
\end{slide}
\begin{slide}{}

The optimality conditions become
\[
\left(
\begin{array}{c}
Z + C - \A^{*}y \\
b-\A(X)\\
\T(ZX-\mu I)\\
\end{array}
\right)
 = 0.
\]
This does not introduce new nonlinearities but does guarantee that we
map between the same spaces. In fact, we can prove that we do not lose
information in the optimality conditions
when we only consider the upper triangular parts.

LEMMA\\
Suppose that $Z \succ 0.$ Then
the linear operator $\T_Z(\cdot) := \T (Z  \cdot )$ is a one-one
mapping on  $\Sn$. In addition, if
$\T_Z(X) = \mu I$, for some $X=X^T$, then $ZX=\mu I.$

\end{slide}
\begin{slide}{}
The Newton direction is obtained from
solving the system
\begin{eqnarray*}
        \Delta Z-\A^{*}(\Delta y)        &=&  -F_{d}  \label{shnewt1}\\
        -\A(\Delta X)                     &=&  -F_{p}  \label{shnewt2}\\
      \T(Z\Delta X+\Delta Z X)             &=&  -\T(Z F_{Z,X}) \label{shnewt4}.
\end{eqnarray*}
Equivalently,
\[
 \left[
\begin{array}{ccc}
0 & -\A^* & I \\
-\A & 0 & 0 \\
\T_Z (\cdot) & 0 & 
\T(\cdot X)
\end{array}
\right]
\left( \begin{array}{c}
    \Delta X \\
    \Delta y \\
\Delta Z \\
            \end{array} \right)
=
\left( \begin{array}{c}
     -F_d\\
     -F_p\\
     -\T_Z(X)+\mu I \\
            \end{array} \right).
\]
\end{slide}
\begin{slide}{}

\begin{center}
MORE LINEARITY\\
 SPEEDS UP
 NEWTON'S METHOD\\
    but\\
results in overdetermined system
\end{center}

\end{slide}
\begin{slide}{}

\subsubsection{Properties of the Gauss-Newton Directions}
\label{sec:properties}
{\bf Well-Defined}
\begin{lemma}\label{GNfullrank}
  If $\A$ is surjective, the Gauss-Newton direction
  exists and is unique for all $X\succ 0$,
  $Z\succ 0$.  (The Jacobian is full-rank.)
\end{lemma}
result requires only positive definite $X$ and $Z$
in contrast to other directions


\end{slide}
\begin{slide}{}

{\bf nonsingularity of Jacobian holds in the limit}\\
\begin{lemma}\label{fullrank0}
  If $\A$ is surjective and the optimal primal-dual solution
  $(X^*,y^*,Z^*)$ is unique and strictly complementary, then
  ${\F^{\prime}_\mu}{(v^*)}$ at $\mu=0$ is nonsingular.
\end{lemma}

\end{slide}
\begin{slide}{}

{\bf Merit function - Descent direction}
  \begin{eqnarray}
  \varphi(v) &:=& \half \inner{F_\mu(v)}{F_\mu(v)} \label{eq:phi}\\
  &=&
  \half\inner{F_d}{F_d}+\half\inner{f_p}{f_p}+\half\inner{F_c}{F_c}\\
  &=:&\varphi_d(v) + \varphi_p(v) + \varphi_c(v).\label{varphidpc}
  \end{eqnarray}

\begin{lemma}\label{gndescent}
  The Gauss-Newton direction $dv$,
  is a strict descent direction for the merit function $f_\mu(v) = \half
 \inner{F_\mu(v)}{F_\mu(v)}$ if and only if $F_\mu(v)$ is not
  perpendicular to the range of ${\F^{\prime}_\mu}{(v)}$.
\end{lemma}



\end{slide}
\begin{slide}{}

\begin{center}
Summary
\end{center}
{\bf Well-Defined}\\
{\bf nonsingularity of Jacobian holds in the limit}\\
{\bf Merit function - Descent direction}\\

 if $\A$ is surjective (and we may assume it is), the
Gauss-Newton direction is a strict descent direction until
stationarity of the merit function is attained.\\
Moreover, under
uniqueness of the optimal solution and strict complementarity,
(conditions that hold generically \cite{PatTun:97}), the Gauss-Newton
system minimizes the risk of ill-conditioning as we approach the
optimal solution.



\end{slide}
\begin{slide}{}

\subsubsection{Conditioning of the Jacobian}
equivalent form of the (overdetermined) optimality system
  \begin{eqnarray}
\mbox{AHO} \quad \left\{
  \begin{array}{rcl}
    \A^*(dy)+dZ &=& -(\A^*(y)+Z-X)  \label{expand_gn1},\\
    \A(dX) &=& -(\A(X)-b), \\
    H(ZdX + dZX) &=& -H(ZX-\mu I), 
  \end{array} 
\right.\\
 \mbox{GN uses extra info}\quad 
                   K(ZdX + dZX) = -K(ZX-\mu I),\label{expand_gn4}
  \end{eqnarray}
where
\[
\begin{array}{rclr}
    H(M) &:=& \frac{1}{2}[M+M^T],&\quad \mbox{(symmetric part);} 
    \label{symm_op}\\
    K(M) &:=& \frac{1}{2}[M-M^T],&\quad \mbox{(skew-symmetric part).}
    \label{skew_op}
\end{array}
\]

\end{slide}
\begin{slide}{}

matrix formulation
\beq\label{aho-gn}
  J_{gn} d = P
  \left[ 
    \begin{array}{c}
      J_{aho} \\ J_{k}
    \end{array}
  \right] d = -
  \left[
    \begin{array}{c}
      F_{s} \\ F_{k}
    \end{array}
  \right],     
\eeq
permutation of the rows $P$\\
$J_{gn}$ - Jacobian of the Gauss-Newton system\\
$J_{aho}$ - Jacobian of the symmetric (AHO) system\\
$J_{k}$ - skew-symmetric component
\begin{eqnarray}
  \bar n &:=& t(n)+m+t(n),\\
  \bar m &:=& t(n)+m+n^2,\\
  r &:=& \bar m - \bar n = t(n)-t(n-1).
\end{eqnarray}
$J_{gn}$ is $(\bar m \times \bar n)$, $J_{aho}$ is
$(\bar n \times \bar n)$, and $J_{k}$ is $(r \times \bar n)$. 



\end{slide}
\begin{slide}{}


ordering of the singular values
\begin{displaymath}
  \sigma_{\max}(J) := \sigma_1(J) \ge \sigma_2(J) \ge \ldots \ge
  \sigma_n(J) =: \sigma_{\min}(J).
\end{displaymath}
\begin{lemma}
  The singular values of $J_{gn}$ and of $J_{aho}$ are related by
  $\sigma_k(J_{gn}) \ge \sigma_k(J_{aho}) \ge
  \sigma_{k+t(n-1)}(J_{gn})$ for $1\le k\le \bar n$.
\end{lemma}
\bpr Follows from {\em interlacing} of singular values.
\cite{HoJo:91}. \epr

Gauss-Newton Jacobian is no closer to singularity than the AHO Jacobian is.

\begin{lemma}
  The largest singular value of the Gauss-Newton Jacobian is bounded
  by $\sigma_{\max}(J_{gn}) \le \sqrt{\sigma_{\max}^2(J_{aho}) +
    \sigma_{\max}^2(J_{k})}$.
\end{lemma}



\end{slide}
\begin{slide}{}


Gauss-Newton system is nonsingular and has a bounded condition number
as we approach the optimal solution. 
(as with AHO direction; but not with NT nor HKM nor original Newton system)

(assuming backward-stable algorithm) the (numerical) direction $\widehat{dv}$
is the exact solution to a nearby problem
\cite{MR2000j:90078}
\begin{displaymath}
  (J_{s} - \delta J_{s}) \widehat{dv} = -(f + \delta f),
\end{displaymath}
with bounded relative error
\beq\label{errorAHO}
  \frac{\|\widehat{dv} - dv\|}{\|dv\|} \le
  \frac{\kappa(J_s)}{1-\kappa(J_s)\frac{\|\delta J_s\|}{\|J_s\|}}
  \Bigl( \frac{\|\delta J_s\|}{\|J_s\|}+ \frac{\|\delta f\|}{\|f\|}\Bigr).
\eeq


\end{slide}
\begin{slide}{}

\begin{lemma}\label{GNeqHKM}
  Suppose that $\A$ is surjective, that $X,y,Z$ is on the central
  path with $F_d=0, F_p=0$ and $ZX-\mu I = 0, ~ \mu > 0$. Suppose that
  the new target for the barrier parameter is $\tau \mu,$ where $0 <
  \tau < 1$. Then the Gauss-Newton direction from 
  coincides with the HKM direction.
\end{lemma}
This is the only case where the Gauss-Newton direction coincides with
other directions of the Monteiro-Zhang family.


\end{slide}
\begin{slide}{}

\subsubsection{Invariance}

\begin{lemma}
  (Q-scale invariance) Let $(dX,dy,dZ)$
  be the Gauss-Newton direction obtained at point
  $(X,y,Z)\in\Snpp\times\Rm\times\Snpp$.  Consider the scaled
  primal-dual pair obtained using
  orthogonal $P$. Then the scaled vector $(PdXP^T,dy,P^{-t}dZT^{-1})$ is
  the Gauss-Newton direction at the iterate $(\widetilde X,\widetilde
  y, \widetilde Z)$ for the scaled problem.
\end{lemma}

\begin{thm}
  The Gauss-Newton direction, at any point $v_c\in \V$ is invariant
  under affine transformation of the variables $w=H(v)+h$ where $H:\V
  \rightarrow \V$ is nonsingular.
\end{thm}



\end{slide}
\begin{slide}{}


\subsubsection{Convergence of Gauss-Newton Method}

semidefinite program pair
\[
  \minpgm{{}}{\inner{C}{X}}{\A(X) = b, X \succeq 0},
\]
\[
  \maxpgm{}{\inner{b}{y}}{\A^*(y)+Z = C, Z \succeq 0}.
\]


merit function: (combined norms infeasibility and complementarity)
\beq
  \label{eq:phimu}
  \varphi(v,\mu) := \half \inner{F_\mu(v)}{F_\mu(v)}.
\eeq


\end{slide}
\begin{slide}{}


  {\bf Generic Gauss-Newton based interior-point code}
\begin{enumerate}
  \item
  \begin{enumerate}
    \item[$\bullet$] Given $\epsilon>0$ \hfill{\em Tolerance}
    \item[$\bullet$] $k := 1; X^{(k)}:= I; Z^{(k)}:= I; y^{(k)}:= 0;
    \mu^{(k)}:=\frac{\inner{Z^{(k)}}{X^{(k)}}}{n}=1;$ 
    \item[$\bullet$]{\bf WHILE}~~{$\max\left\{\varphi(v^{(k)},\mu^{(k)}\right\} > \epsilon$}
  \begin{enumerate}
    \item[$\bullet\bullet$] $d^{(k)} = - 
      {F^{\prime}_\mu(v^{(k)})}F_\mu(v^{(k)});$\hfill
                  {\em Gauss-Newton direction}
    \item[$\bullet\bullet$] $\alpha^{(k)} :=
    \mbox{LineSearch}(v^{(k)},d^{(k)},\mu^{(k)});$\hfill{\em Decrease
    $\varphi$}
    \item[$\bullet\bullet$] $\mu^{(k+1)} :=
    \mbox{TargetSelect}(v^{(k)},d^{(k)},\alpha^{(k)},\mu^{(k)});$\\
      ~~~~~~~~~~~\hfill{\em $\mu^{(k+1)} = \tau\mu^{(k)}, 0\le \tau <1$}
    \item[$\bullet\bullet$] $v^{(k+1)} := v^{(k)} + \alpha^{(k)}
    d^{(k)};$\hfill{\em New iterate}
    \item[$\bullet\bullet$] $k := k+1;$
   \end{enumerate}
    {\bf ENDWHILE}
  \end{enumerate}
\end{enumerate}



\end{slide}
\begin{slide}{}


{\bf Now: assumptions for convergence analysis}\\
  \begin{itemize}
  \item $\exists v^0$ satisfying centrality condition 
  \item operator $\A$ is surjective.
  \item optimal primal-dual pair is unique and satisfies strict
    complementarity (i.e. $Z+X \succ 0$).
  \end{itemize}

-algorithm approximately follows the
central path by attempting to solve $F_\mu(X,y,Z)=0$ for decreasing
values of $\mu$\\


\end{slide}
\begin{slide}{}

-major difference: the scalar $\mu$ is not updated using the iterates,
but is reduced by a factor $\tau<1$ at every step ($\mu \leftarrow \tau
\mu$);\\  
no attempt is made to dampen the step to maintain the iterates
within the cone of positive definite matrices. (The algorithm only
maintains the weaker full rank condition on the Jacobian.)

\begin{lemma}
  The operator $F_{\tau \mu}^\prime(s)$ is Lipschitz continuous 
with constant 1.
\epr
\end{lemma}

\end{slide}
\begin{slide}{}

\begin{thm} \label{stdcvg}
  Let $\sigma_{\min}$ and $\sigma_{\max}$ be, respectively, the
  smallest and largest singular value of $F_{\tau \mu}^\prime(s_{\tau\mu})$.
  There is a $\delta > 0$ such
  that for all $s_c$ such that $\|s_c - s_{\tau\mu}\| < \delta$, the
  Gauss-Newton step
  \begin{displaymath}
    s_+ = s_c - F_{\tau \mu}^\prime(s_c)^\dagger F_{\tau \mu}(s_c)
  \end{displaymath}
  is well-defined and converges to $s_{\tau\mu}$ at a rate such that
  \begin{displaymath}
    \|s_+ - s_{\tau\mu}\| \le \frac{1}{\sigma_{\min}} \|s_c -
s_{\tau\mu}\|^2.
  \end{displaymath}
  Moreover, we can choose $\delta$ as long as $\delta < \frac
  {\sigma_{\min}}{2}$.
\epr
\end{thm}


\end{slide}
\begin{slide}{}

\begin{cor}\label{gndecrease}
  Let $\sigma_{\min}$ and $\sigma_{\max}$ be, respectively, the
  smallest and largest singular value of $F_{\tau\mu}^\prime(s_{\tau\mu})$.
  There is a $\delta > 0$ where
  for all $s_c$ such that $\|s_c - s_{\tau\mu}\| < \delta$,
  \begin{displaymath}
    \|F_{\tau\mu}(s_+)\| \le \frac{1}{2} \|F_{\tau\mu}(s_c)\|.
  \end{displaymath}
  Moreover, we can choose any $\delta$ such that
  \beq
    \label{eq:deltacor}
  \delta <
  \frac{\sigma_{\min}^2}{8 \sigma_{\max} }.
\eeq
\epr
\end{cor}


\end{slide}
\begin{slide}{}

general idea:\\
 from an iterate $s_k$ ,
``close enough'' to $s_\mu$, we can choose a target on the central
path $s_{\tau\mu}$ in such a way that the next iterate $s_{k+1}$,
obtained from the Gauss-Newton direction, is now ``close enough'' to
$s_{\tau\mu}$ for the process to be repeated.


\end{slide}
\begin{slide}{}


First we estimate the distance between
two points on the central paths in terms of the required radius of
convergence.
\begin{lemma}\label{mu2taumu}
  Let $\sigma_{\min}$ and $\sigma_{\max}$ be, respectively, the
  smallest and largest singular value of $F_\tau^\prime(s_{\tau\mu})$.
  Let $s_\mu$ and $s_{\tau\mu}$ be the optima on the central path.
\begin{enumerate}
\item
\label{item:part1}
If we choose $0 < \tau < 1$ such that
  \beq
    \label{eq:tauvalue}
    1-\tau \le \frac{\sigma_{\min}^2}{8\sqrt{n}\mu},
  \eeq
  then
  \beq \label{eq:halfrad}
    \|s_\mu - s_{\tau\mu}\| \leq \frac 12 
    \left( \frac {\sigma_{\min}}{2}  \right),
  \eeq
  which implies $s_\mu$ is within half of the radius of quadratic
  convergence of $s_{\tau\mu}$.
\item
  \label{item:part2}
  If we choose $0 < \tau < 1$ such that
  \beq
    \label{eq:tauvalueforF}
    1-\tau \le \frac{
      \sigma_{\min}^3}{32 \sqrt{n}\mu
      \sigma_{\max} },
  \eeq
  then
  \beq \label{eq:halfradforF}
    \|s_\mu - s_{\tau\mu}\| \leq \frac 12 
    \left( \frac {\sigma_{\min}^2}{8
        \sigma_{\max}}  \right).
  \eeq
  In this case $s_\mu$ is within half of the radius of guaranteed constant
  decrease of the merit function in (\ref{eq:deltacor}) in Corollary
  \ref{gndecrease}.
\end{enumerate}
\epr
\end{lemma}

\end{slide}
\begin{slide}{}

We now estimate the distance to the new target after a Gauss-Newton step.
\begin{lemma}
  \label{lem:totarget}
  Let $\sigma_{\min}$ and $\sigma_{\max}$ be, respectively, the
  smallest and largest singular value of $F_\tau^\prime(s_{\tau\mu})$.
  Let $s_\mu$ and $s_{\tau\mu}$ be optima on the central path. Suppose that
  the point $s_c$ is well-centered in the sense that
  \begin{eqnarray*}
    \|s_\mu - s_c\| &\le& 
    \min\left\{\frac{\sigma_{\min}}{4},\frac{\sigma_{\min}}{16\sigma_{\max}}\right\},
  \end{eqnarray*}
  and we choose $\tau$ to satisfy
  \begin{equation}
    \label{eq:tauvalue2}
    0 < \tau < 1, \quad
    1-\tau \le 
    \min \left\{\frac{\sigma_{\min}^2 }{8 \sqrt{n}\mu}, \frac{\sigma_{\min}^3 }{32 \sqrt{n}\mu}\right\},
  \end{equation}
  as in Lemma \ref{mu2taumu}. Then, after one Gauss-Newton step, the
  new point $s_+$ will be within half the radius of convergence of
  $s_{\tau\mu}$, i.e.
  \begin{eqnarray}
    \|s_{\tau\mu} - s_+\| &\le& \frac{\sigma_{\min} }{4}.
  \end{eqnarray}
  Moreover, the merit function is reduced
  \begin{equation} \label{eq:meritred}
    \|F_{\tau}(s_+)\| \leq \frac 12 \|F_{\tau}(s_c)\|.
  \end{equation}
\epr
\end{lemma}

\end{slide}
\begin{slide}{}

Main Convergence Theorem:
\begin{thm}
  \label{thm:mainthm}
  Suppose that we are given a tolerance $\epsilon > 0$, an initial
  barrier parameter $\mu_0 > \epsilon$, and $Z_0,X_0 \in\Snpp$ such
  that $s_0=(X_0,y_0,Z_0)$ is a well-centered starting point: $s_0$ is
  within half the quadratic convergence radius of $s_{\mu_0}$ in
  Theorem \ref{stdcvg},
  \begin{equation}\label{s0}
    \|s_{\mu_0} - s_0\| \leq \frac 12 \left( \frac {\sigma_{\min}} {2} \right).
  \end{equation}
  Suppose moreover that $s_0$ is within half the radius for guaranteed
  constant decrease of the merit function given in Corollary
  \ref{gndecrease},
  \begin{displaymath}
    \|s_{\mu_0}-s_0\|\leq\frac 12\left(\frac{\sigma_{\min}^2}{8\sigma_{\max}}\right),
  \end{displaymath}
  where $0< \sigma_{\min}$ (respectively $\sigma_{\max}$) is smaller
  than the smallest (respectively larger than the largest) singular
  value of $F_{\omega \mu_0}^{\prime}(s_{\omega \mu_0})$, for all
  $\frac {\epsilon}{\mu_0} < \omega < 1$.  
  
  If we choose $\tau$ ({\em small}) satisfying (\ref{eq:tauvalue2}) in
  Lemma \ref{lem:totarget}, i.e.
  \begin{displaymath}
    \alpha = \min \left\{ 
      \frac {\sigma^2_{\min}}{8\sqrt{n}\mu_0}, \,
      \frac {\sigma^3_{\min}}{32\sqrt{n}\mu_0} 
    \right\},
  \end{displaymath}
  and $\tau \geq \max\left\{0,1-\alpha\right\},\,0<\tau<1$, then
  the GN Algorithm converges to $\bar{s}$, $\epsilon$-optimal
  in the following sense,
  \begin{displaymath}
    \tau^k\mu_0 \le \epsilon, \quad
    \|F_{\tau^k \mu_0}(\bar{s})\| \leq \epsilon, \quad 
               \|\bar{s}-s_{\tau^k\mu_0}\| \le 2 \sigma_{\min} \epsilon;
  \end{displaymath}
  and the number of iterations, $k$, depends on $\tau$:
\begin{enumerate}
%%%%%%%%%%%%%%%%%%%%% Simplification since we are using big-Oh
%%%%%%%%%%%%%%%%%%%%% You can replace by the commented out part if you
%%%%%%%%%%%%%%%%%%%%% want 
% \item 
%\begin{enumerate}
% \item 
%   \begin{equation}
%     \label{eq:iterations1}
%     {\cal O}\left(\max \left\{
%         \left(
%           \frac 
%           {\log \left(\frac{\|F_{\mu_0}(s_0)\|}{\epsilon} \right)}
%           {\log {2} } +1
%         \right),\,\,\,
%         2\left( \frac {\log        \frac {\mu_0\sqrt{n}}{\epsilon} }
%           {\log 2}+1 \right),\,\,\,
%         \left( \frac {-\log \frac {\epsilon}{\mu_0}}{\log 2} \right)
%       \right\} \right)  
%   \end{equation}
%   iterations, if $0<\tau \leq \frac 12$;
% \item
%   \begin{equation}
%     \label{eq:iterations2}
%     {\cal O}\left(\max \left\{
%         \left(
%           \frac 
%           {\log \left(\frac{\|F_{\mu_0}(s_0)\|}{\epsilon} \right)}
%           {\log {2} } +1
%         \right),\,\,\,
%         \left( \frac {\log   
%             \frac 
%             {(2\tau-1)\epsilon}{(1-\tau)2\mu_0\sqrt{n} }   }
%           {\log \tau}  \right),\,\,\,
%         \left( \frac {\log \frac {\epsilon}{\mu_0}}{\log \tau} \right)
%       \right\} \right)  
%   \end{equation}
%   iterations, if $\frac 12<\tau < 1$.
% \end{enumerate}
%%%%%%%%%%%%%%%%%%%%%%%%%%%  End of commented out region
%%%%%%%%%%%%%%%%%%%%%%%%%%%  Begin replacement
\item 
  \begin{equation}
    \label{eq:iterations1}
    {\cal O}\left(\max \left\{
          \log \left(\frac{\|F_{\tau \mu_0}(s_0)\|}{\epsilon} \right)
        ,\,\,\,
        \log \left( \frac {\mu_0\sqrt{n}}{\epsilon} 
           \right)
      \right\} \right)  
  \end{equation}
  iterations, if $0<\tau \leq \frac 12$;
\item
  \begin{equation}
    \label{eq:iterations2}
    {\cal O}\left(\max \left\{
          {\log \left(\frac{\|F_{\tau \mu_0}(s_0)\|}{\epsilon} \right)}
        ,\,\,\,
        \left( \frac {\log   
            \frac 
            {(2\tau-1)\epsilon}{(1-\tau)2\mu_0\sqrt{n} }   }
          {\log \tau}  \right),\,\,\,
        \left( \frac {\log \frac {\epsilon}{\mu_0}}{\log \tau} \right)
      \right\} \right)  
  \end{equation}
  iterations, if $\frac 12<\tau < 1$.
\end{enumerate}
\epr
\end{thm}

\end{slide}
\begin{slide}{}



{Asymptotic Convergence}
We are also able to specialize a standard result pertaining to the
asymptotic convergence rate of the Gauss-Newton method. 
\begin{thm} (Superlinear Convergence)
  Assume a primal-dual pair with a unique, strictly complementary,
  optimal solution, denoted $v^*$. Then for each $c \in (1, \infty)$,
  there is an $\epsilon > 0$ such that, from $v^{(0)}$ satisfying
  $\|v^{(0)}-v^*\| < \epsilon$, the sequence generated by
  the Gauss-Newton method converges to $v^*$ and obeys
  \begin{equation}
    \label{eq:ds1}
    \|v^{(k+1)} - v^* \| \le \frac{c\alpha}{2\sigma_{\min}^2}\|v^{(k)} - v^* \|^2.
  \end{equation}
  where $\|J(v)\| \le \alpha$ in the $\epsilon$-neighborhood of $v^*$
  and where $\sigma_{\min}$ represent the smallest singular value of
  $J(v^*)$.
\end{thm}

This last result suggests that, whatever upper-level algorithm we use,
if we use the Gauss-Newton as the search direction, then as soon as we
can estimate $\sigma_{\min}$ accurately, we compute the radius of
superlinear convergence and we can set the barrier parameter to zero,
ignore the cone constraint and let the algorithm converge to the
optimal solution.



\end{slide}
\begin{slide}{}

\subsubsection{Numerical Tests with GN}

\tiny{
  \begin{center}
    \begin{tabular}{||c|c|c|c||}\hline
      Iter & $\|F_d\|$ & $\|f_p\|$ & $\inner{Z}{X}/n$\\
      \hline\hline
      1 & +8.900e+01 & +4.175e+01 & +8.749e-01 \\ 
      2 & +5.433e+01 & +3.545e+01 & +8.580e-01 \\
      3 & +1.503e+01 & +1.759e+01 & +6.269e-01 \\
      4 & +3.066e+00 & +3.465e-03 & +3.841e-01 \\
      5 & +2.131e-02 & +4.408e-04 & +1.278e-01 \\
      6 & +5.573e-03 & +8.302e-05 & +2.873e-02 \\
      7 & +1.042e-04 & +3.153e-06 & +1.367e-02 \\
      8 & +1.611e-05 & +2.913e-07 & +7.862e-04 \\
      9 & +7.275e-08 & +1.620e-09 & +8.680e-05 \\
      10 & +4.473e-13 & +2.041e-14 & +1.812e-06 \\
      11 & +1.017e-14 & +2.800e-15 & +3.784e-08 \\
      12 & +1.030e-14 & +3.426e-15 & +7.902e-10 \\
      13 & +1.318e-14 & +5.184e-15 & +1.650e-11 \\
      14 & +1.138e-14 & +4.835e-15 & +4.964e-14 \\
      \hline
    \end{tabular}
  \end{center}
    {\bf One instance of well-conditioned problem. $n=15,m=30$}
}

\end{slide}
\begin{slide}{}

\tiny{
  \begin{center}
    \begin{tabular}{|c|cccccccccc|}\hline
      &\multicolumn{5}{c|}{$-\log_{10}\max\{\|F_d\|,\|f_p\|\}$} & 
      \multicolumn{5}{c|}{$-\log_{10}\inner{Z}{X}/n$}\\ \hline
      & AHO & HKM & NT & GT & GN & AHO & HKM & NT & GT & GN\\ \hline
      random &      14.7 & 9.1 & 10.6 & 9.5 & 13.3 & 12.0 & 10.7 & 5.5 & 12.8 & 14.4   \\ 
      norm min. &   15.0 & 9.9 & 10.9 & 15.3 & 14.8 & 13.8 & 12.7 & 10.1 & 14.4 & 14.9   \\
      Maxcut&       15.7 & 9.6 & 10.6 & 14.8 & 15.8 & 14.4 & 12.4 & 9.3 & 14.4 & 15.5   \\
      Lov{\`a}sz $\theta$&  14.7 & 9.2 & 9.2 & 13.9 & 14.8 & 14.2 & 13.7 & 12.8 & 14.1 & 15.3   \\
      \hline
    \end{tabular}
  \end{center}
  {\bf Accuracy of solutions on SDPT3 test problems. Average
      of one hundred random instances.}
}

\end{slide}
\begin{slide}{}


  \begin{center}
    \begin{tabular}{|c|cccccc|}\hline
      & \multicolumn{3}{c}{AHO}& \multicolumn{3}{c|}{GN}\\
      $r,n,m$ & iter. & Infeas. & Gap  & iter. & Infeas. & Gap \\ \hline
      3,10,9 & 18  &  14.2  &  15.1  &  13  &  14.3  &  15.4\\
      6,20,24 & 22  &  12.1  &  14.6 &   17  &  13.8  &  16.3\\ \hline
    \end{tabular}
  \end{center}
  {\bf Ill-conditioned problems' solutions. Average of fifty
      random instances.}

\end{slide}
\begin{slide}{}


  \begin{center}
    \begin{tabular}{|c|c||c|c|c|c|}\hline
      $n$ & $m$ & iter & $\|F_d\|$& $\|f_p\|$&$\inner{Z}{X}$\\\hline
      10 & 11 & 24 &9.296152e-13& 5.594466e-11& 5.842689e-07\\\hline
      20 & 21 & 25 &2.260052e-12& 4.882512e-15& 7.906645e-07\\\hline
      30 & 31 & 22 &4.211995e-12& 9.367561e-15& 2.190712e-06\\\hline
      40 & 41 & 23 &1.816565e-12& 7.334760e-15& 2.301071e-06\\\hline
    \end{tabular}
  \end{center}
  {\bf Gruber-Rendl ill-posed Problems with
      $\alpha=10^{-7}$ and accuracy set to $10^{-5}$.}

(average iterations for G-R was 115; $\alpha = 0$ means no primal Slater
point)


\end{slide}
\begin{slide}{}

  \begin{center}
    \begin{tabular}{|c|c||c|c|c|c|}\hline
      $n$ & $m$ & iter & $\|F_d\|$& $\|f_p\|$&$\inner{Z}{X}$\\\hline
      10  & 11  & 23   & 2.377587e-12 & 4.076899e-08 &
9.096929e-06\\\hline
      20  & 21  & 22   & 2.479800e-12 & 1.516398e-08 &
1.050569e-06\\\hline
      30  & 31  & 25   & 5.153522e-12 & 5.729514e-08 &
1.416319e-06\\\hline
    \end{tabular}
  \end{center}
  {\bf  with $\alpha=0$
      and accuracy set to $10^{-5}$.}

~~\\

  \begin{center}
     \begin{tabular}{|c|c||c|c|c|c|}\hline
      $n$ & $m$ & iter & $\|F_d\|$& $\|f_p\|$&$\inner{Z}{X}$\\\hline
      50 & 51 & 32 & 3.467449e-12 & 6.469784e-15 & -1.233752e-14\\\hline
   \end{tabular}
  \end{center}
   {\bf  with $\alpha=0$ and
      increased accuracy.}




\end{slide}
\begin{slide}{}


  \begin{center}
    \begin{tabular}{|c|c||c|c|c|c|}\hline
      $n$ & $m$ & iter & $\|F_d\|$& $\|f_p\|$&$\|\inner{Z}{X}\|$\\\hline
      10  & 9  & 7 &  8.015765e-11 & 8.009760e-11 & 5.080324e-06\\\hline
      20  & 19 & 8 &  9.000070e-11 & 9.001461e-11 & 1.743560e-06\\\hline
      30  & 28 & 8 &  1.272965e-10 & 9.000636e-11 & 2.615344e-06\\\hline
      40  & 15 & 5 &  2.593612e-09 & 9.999999e-11 & 1.651607e-07\\\hline 
      40  & 30 & 5 &  2.642280e-09 & 1.000001e-10 & 3.821364e-07\\\hline
      40  & 39 & 5 &  2.231644e-09 & 9.999997e-11 & 5.396936e-07\\\hline
      50  & 49 & 5 &  2.928152e-09 & 9.999996e-11 & 6.840107e-07\\\hline
    \end{tabular}
  \end{center}
    {\bf no primal or dual Slater points: accuracy set to $10^{-5}$.}

(G-R needed average of 52 iterations)

\end{slide}
\begin{slide}{}

\subsubsection{Solving SDPs using NLP}
Though the current popular approaches for solving SDP use extensions
of interior-point methods from LP (successful approaches), there are
several that use NLP techniques, e.g. Monteiro-Zhang... or Vanderbei...
details????

However, we use the primal-dual framework with known results from NLP to
derive the Gauss-Newton method.

Provide convergence results based on standard NLP approaches (redo KW
paper using $\cos \theta$ bounded below and Wolfe conditions)


\end{slide}
\begin{slide}{}


\subsection{Exploiting Sparsity}
\subsubsection{Special Structure}
\subsubsection{Bundle Trust Methods}
\subsubsection{Potential Function Methods}


\label{endsect:algorithms}
\end{slide}
\begin{slide}{}

\bs{Applications}
\label{sect:appl}
\subsection{Outline}

\begin{enumerate}
\item[$\bullet$]
Quadratic Boolean Programming
\item[$\bullet$]
Quadratically Constrained Quadratic Programs
\item[$\bullet$]
Trust Region Algorithms for NLP
\item[$\bullet$]
Sequential Quadratic Programming for NLP
\item[$\bullet$]
Historical Notes

\end{enumerate}



\end{slide}
\begin{slide}{}

\subsection{Quadratic Boolean Programming}
\label{sect:QBP}


\subsubsection{Outline}
\begin{enumerate}

\item[$\bullet$]
Tractable Relaxations
\item[$\bullet$]
Strengthened SDP relaxations
\item[$\bullet$]
Strengthened Bounds for Max-Cut
\item[$\bullet$]
Historical Notes

\end{enumerate}




\end{slide}
\begin{slide}{}

\subsubsection{Tractable Relaxations of Max-Cut}
(quadratic boolean programming)
\[
(MCQ)~~~  \begin{array}{c}
   \mu^*:= \max\limits_{x \in \FF} q_0(x) \quad ( : = x^TQx -2 c^Tx).
\end{array}
\]
where $\FF=\left\{\pm 1\right\}^n$\\
\underline{\hspace{90mm}}\\

perturbing diagonal of $Q$ on $\FF:$
\[
\begin{array}{rcl}
q_u(x)& : = &x^T(Q+\Diag(u))x -2 c^Tx - u^Te\\
      &~=&q_0(x),~~\forall x \in \FF,
\end{array}
\]
where $e$ vector of ones.

\end{slide}
\begin{slide}{}
{\bf Bound 0:}
trivial bound from diagonal perturbations:
\[
  \mu^* \leq f_0 (u) := \max_x q_u(x).
\]
The function $f_0$ can take on the value $+\infty.$ 
Let 
\[S:=\left\{u:u^Te=0, Q+\Diag(u)\preceq 0\right\}.
\]
Then:
\[
\begin{array}{|c|}
\hline\\
  \mu^* \leq B_0 := \min\limits_u  f_0(u) \\
    \left(= \min\limits_{u^Te=0}  f_0(u),
       \mbox{ if } S \neq \emptyset   \right).  \\
~\\
\hline
\end{array}
\]
Using the hidden semidefinite constraint:
\[
  \mu^* \leq B_0 = \min\limits_{Q+\Diag(u) \preceq 0}  f_0(u).
\]

\end{slide}
\begin{slide}{}
{\bf Bound 1:}
relax the feasible set to
the sphere of radius $\sqrt{n}$ (tractable trust region subproblem, TRS):
\[
\mu^* \leq f_1(u):= \max_{|| x ||^2 =n} ~ q_u(x)
\]
and
\[
\begin{array}{|c|}
\hline\\
  \mu^* \leq B_1 :=\min\limits_u f_1(u).\\
~~\\
\hline
\end{array}
\]


The inner maximization problem is called a
trust region subproblem and is tractable.
This bound provides the central tool in the proofs of equivalence.

\end{slide}
\begin{slide}{}
{\bf Bound 2:} box constraint:
\[
\mu^* \leq f_2(u):= \max_{| x_i | \leq 1} ~ q_u(x).
\]
add the semidefinite constraint to make bound
tractable.
\[
  \mu^* \leq \min\limits_{u} f_2(u)
\]
and
\[
\begin{array}{|c|}
\hline\\
  \mu^* \leq B_2 :=\min\limits_{Q+\Diag(u) \preceq 0} f_2(u).\\
~~\\
\hline
\end{array}
\]


\end{slide}
\begin{slide}{}
{\bf Bound $B_1^c$:} lift to eigenvalue bound:

\[
 Q^c := \left[ \begin{array}{cc}
   0 &  - c^T \\
  - c &Q    \end{array}  \right]
\]
\[
 q^c_u(y) :=   y^T( Q^c+\diag(u))y-u^Te
\]
\[
  \mu^* \leq f_1^c(u) := \max_{||y||^2=n+1} q^c_u(y)
\]
where
\[
   \max_{||y||^2=n+1} q^c_u(y)
   = (n+1) \lambda_{\max} (Q^c + \diag (u) ) - u^Te
\]
\[
\begin{array}{|c|}
\hline\\
  \mu^* \leq B_1^c :=  \min\limits_{u} f_1^c(u).\\
~~\\
\hline
\end{array}
\]

Similarly, equivalent bnds $B_0^c$
and homog. bnds for other models.


\end{slide}
\begin{slide}{}
{\bf Bound $B_3$:} SDP bound:

After homogenization ( $c=0$), use
\[ x^TQx = \tr x^TQx = \tr Qxx^T\]
and, for $x \in \FF,$ $y_{ij} = x_ix_j$ defines a
symmetric, rank one, positive semidefinite matrix $Y$ with diagonal elements 1.
Relax the rank one condition.
\[
\begin{array}{|ccc|}
\hline
&&\\
       B_3 :=&\max &\tr QY \\
 &  \mbox{~subject to~} &\diag(Y) = e \\
        && Y \succeq 0.\\
&&~\\
\hline
    \end{array}
\]



\end{slide}
\begin{slide}{}
Summary: ref. Polyak-Rendl-Wolkowicz (PRW) 1995.
(without restrictions e.g. $u^Te=0$)


\[
\begin{array}{|rcl|}
\hline
&&~\\
 B_0 &=&\min\limits_u \max\limits_x q_u(x)\\
 B_1 &=&\min\limits_u \max\limits_{x^Tx=n} q_u(x)\\
 B_2 &=&\min\limits_u \max\limits_{-1 \leq x_i \leq 1} q_u(x)\\
 B_3 &=&\max \{\tr Q^cY : \diag(Y) = e,~ Y \succeq 0. \}\\
 B_1^c &=&\min\limits_u \max\limits_{y^Ty=n+1} q^c_u(y)\\
&&~\\
\hline
\end{array}
\]


\end{slide}
\begin{slide}{}

Now replace $\pm 1$ constraints with $x_i^2=1, \forall i.$
\[
\begin{array}{ccc}
       (P_E)&\max & q_0(x)=x^TQx-2c^Tx \\
 &  \mbox{~subject to~} &
        x_i^2 = 1,~~i=1, \cdots, n.  \\
    \end{array}
\]
$B_L$ denotes Lagrangian relaxation bound.

Following the theme: ``Lagrangian relaxation is best''
\begin{center}
\fbox{
{\bf Theorem } $B_L$ equals all above bounds.
}
\end{center}

 ref PR, PRW/95.\\
(The proofs come from exploiting strong Lagrangian duality of TRS.)


\end{slide}
\begin{slide}{}

\subsubsection{Strengthened SDP Bounds for Max-Cut Problem}

(ref Anjos-Wolkowicz/99 -Illustration of the recipe for SDP relaxation.
based on a
second lifting - motivated by strong duality results that follow
from adding redundant constraints of type
$XX^T$, ref. Anstreicher-Wolkowicz/98 and below)

first lifting procedure
\[ X=xx^T, \quad  \Diag X = e \]
implies second lifting procedure
\[X^2=xx^Txx^T=nX.\]
and provides an equivalent quadratic matrix model for MC

\end{slide}
\begin{slide}{}

MCQ is equivalent to:
\begin{eqnarray*}
\mu^*:=&\max& \tr QX\\
&\mbox{s.t.}&\diag(X)=e\\
&&X \succeq 0\\
&& X \mbox{ is rank 1}.
\end{eqnarray*}
which is equivalent to
\begin{eqnarray*}
\mu^*:=&\max& \tr QX\\
&\mbox{s.t.}&\diag(X)=e\\
&&X^2-nX=0
\end{eqnarray*}
Since $X$ and $X^2$ can be mutually diagonalized.

\end{slide}
\begin{slide}{}


Now add redundant constraints and get:
\begin{eqnarray*}
\mu^*:=&\max& \tr QX\\
&\mbox{s.t.}&\diag(X)=e\\
&&X^2-nX=0\\
&&X \circ X=E
\end{eqnarray*}
where $E$ is the matrix of ones.

Note: $\left(X^2=nX,~\trace X = n\right) \Rightarrow \left(\rank X=1\right)$.
But, nonconvex problem too hard to solve in general.
Need Lagrangian relaxation with vectors dimension $t(n)=n(n+1)/2.$


\end{slide}
\begin{slide}{}

Recipe for SDP relaxations:
\begin{enumerate}
\item   add redundant constraints 
\item   take Lagrangian dual 
\item   homogenize 
\item   use hidden semidefinite constraint to obtain SDP equivalent 
            {\em (check Slater's constraint qualification -
                      strict feasibility)}
\item   take Lagrangian dual again
\item   check Slater's CQ again - project if it fails
\item   delete redundant constraints
\end{enumerate}



\end{slide}
\begin{slide}{}

Notation:\\
(Interesting operators and adjoints
 --fun for me, if for no one else)

~$\bullet$ $S \in \Sn, \quad t(n)= \frac {n(n+1)}2$

~$\bullet$ $s=\svec(S) \in \RR^{t(n)} \quad$ 
vector formed (columnwise) from $S$
ignoring strictly lower triang.

~$\bullet$ $S=\sMat(s) \quad$ inverse of $\svec$

~$\bullet$ $\hMat (v) \quad$ 
is adjoint of $\svec$,
off-diagonal terms are multiplied by a half

~$\bullet$ $\dsvec (S) \quad$ is
adjoint of $\sMat$, operator 
like $\svec$ but off diagonal elements are
multiplied by 2

~$\bullet$ $\sdiag(s):= \diag( \sMat(s))$

~$\bullet$ $\vsMat(s):=\kvec(\sMat(s))$
\[ \bullet \quad 
  \vsMat^*(s)=\dsvec \left( \left(\kmat(v)+\kmat(v)^T\right)/2 \right).
\]


\end{slide}
\begin{slide}{}

start with equivalent program to MCQ:
\beq \label{eq:maxcutstrength}
 {\rm MC2} \begin{array}{ccc}
    \mu^*=
    &\max & \ \tr QX \\
  &\mbox{s.t.}& \diag(X) = e\\
     &&X\circ X=E\\
     &&X^2-nX=0,\\
\end{array}
\eeq
to efficiently apply Lagrangian relaxation and not lose
information from the linear constraint, replace the
constraint with the norm constraint $||\diag(X) - e||^2=0$.
Add $1-y_0^2=0.$

We use: $X=\sMat(x)$ is  a symmetric matrix.

\[
\begin{array}{ccc}
    &\max &  \tr \left(Q\, \sMat(x)\right)y_0 \\
  &\mbox{s.t.}& \sdiag(x)^T \sdiag(x) - 2 e^T \sdiag(x)y_0 
                  + n=0\\
     &&\sMat(x) \circ \sMat(x)=E\\
     &&\sMat(x)^2-n\, \sMat(x)y_0=0\\
     &&1-y_0^2=0.
\end{array}
\]



\end{slide}
\begin{slide}{}

take the Lagrangian dual;
Lagrange multipliers $w,T,S$
\[
 \begin{array}{lcc}
     \mu^*\leq \\
\nu_2^*:=
    \min\limits_{w,T,S}
\max\limits_{x,y_0^2=1}   \tr \left(Q\sMat(x)\right)y_0 \\
  ~~~~+w( \sdiag(x)^T \sdiag(x) - 2 e^T \sdiag(x)y_0 + n)\\
  ~~~~~~~~~~~+\tr T(E-\sMat(x) \circ \sMat(x))\\
  ~~~~~~~~~~~~~ +\tr S((\sMat(x))^2-n\, \sMat(x)y_0).
 \end{array}
\]
move variable $y_0$ into the Lagrangian
without increasing the duality gap since this is a trust region
subproblem

\[
 \begin{array}{rl}
      \nu_2^*=&
    \min\limits_{t,w,S}
\max\limits_{x,y_0}   \tr \left(Q\sMat(x)\right)y_0 \\
 &~+w( \sdiag(x)^T \sdiag(x) - 2 e^T 
                  \sdiag(x)y_0 + n)\\
 &~~+\tr T(\sMat(x) \circ \sMat(x)-E)\\
 &~~+\tr S((\sMat(x))^2-n\, \sMat(x)y_0)\\
 &~~~+ t(1-y_0^2).
\end{array}
\]

\end{slide}
\begin{slide}{}

The inner maximization of the above relaxation is an unconstrained
pure qua\-d\-r\-a\-tic maximization, i.e. 
the optimal value  is infinity unless the Hessian is negative
semidefinite {\bf (hidden constraint)} is which case $x=0$ is optimal.

Therefore we need to evaluate the Hessian of the Lagrangian.


\end{slide}
\begin{slide}{}

Using $Q \sMat(x) = x^T \dsvec(Q),$ and adding a 2 for convenience, we get
the constant part (no Lagrange multipliers) of the Hessian:
\[
 2H_c:=2\pmatrix{
  0 & \frac 12 \dsvec(Q)^T\cr
 \frac 12 \dsvec(Q)  & 0\cr
}.
\]


\end{slide}
\begin{slide}{}


nonconstant part:

Use:

$\dsvec \Diag \diag  \sMat = \sdiag^* \sdiag = \Diag \svec (I)$

rewrite the quadratic forms as follows:

\[
\begin{array}{lll}
\sdiag(x)^T \sdiag(x) = 
     x^T \left(\dsvec \Diag \diag  \sMat\right)x;\\
e^T \sdiag(x) = 
     \left(\dsvec \Diag e  \right)x;\\
~~\\
\tr S (\sMat(x))^2 = \tr  \sMat(x)\, S\, \sMat(x)\\
   ~~~~~  = x^T \dsvec\left(S \sMat(x)\right)\\
     ~~~~~= x^T \left(\dsvec S \sMat\right)x;\\
\mbox{ }\\
     \tr T(\sMat(x) \circ \sMat(x))\\
     ~~= x^T \left\{\dsvec \left(T \circ \sMat(x) \right) \right\} \\
     ~~= x^T \left(\dsvec \left(T \circ \sMat \right)\right) x.
\end{array}
\]


\end{slide}
\begin{slide}{}

use the {\em negative} of the Hessian and split it into four
linear operators with the factor 2:
\[
   \begin{array}{lll}
 2\HH:= 2\HH(w) + 2\HH(T)+ 2\HH(S) + 2\HH(t) \\
  ~~:=2w\pmatrix{
  0 &  (\dsvec \Diag e )^T\cr
  (\dsvec \Diag e ) & -\sdiag^* \sdiag }\\
  ~~~~~+2\pmatrix{
  0 &  0 \cr
   0 & \dsvec \left( T \circ \sMat\right) }\\
  ~~~~~~~~+2\pmatrix{
  0 & \frac 1n\dsvec(S)^T\cr
   \frac 1n\dsvec(S)
        & -\dsvec S  \sMat }\\
  ~~~~~~~~~~~+2t\pmatrix{
  1 & 0\cr
   0 & 0 }.\\
\end{array}
\]



\end{slide}
\begin{slide}{}

The matrix
$\sdiag^* \sdiag \in {\cal S}^{t(n)}$ is diagonal with elements
determined using 
\begin{eqnarray*}
e_i^T \left( \sdiag^* \sdiag \right)e_j
&=&\sdiag(e_i)^T\sdiag(e_j)\\
&=&
\left\{ \begin{array}{cc} 1 & \mbox{if } $i=j=t(k)$\\
               0  &  \mbox{otherwise.}
         \end{array}   \right.
\end{eqnarray*}
Similarly, we find that, for $T=\sum_{ij} t_{ij}E_{ij},$ where the
matrices $E_{ij}$ are the elementary matrices $e_ie_j^T+e_je_i^T,$ we have
\[
\dsvec \left( T \circ \sMat\right)= \sum_{ij} t_{ij}
                  \dsvec \left(E_{ij}\circ \sMat\right).
\]


\end{slide}
\begin{slide}{}

cancel the 2; get 
(equivalent to the Lagrangian dual)
\[
\mbox{MCDSDP2} \quad
  \begin{array}{ccc}
     \nu^*_2 =
    &\min & nw +\tr ET+\tr 0S  +t\\
  &\mbox{s.t.}& \HH(w,T,S,t) \succeq H_c\\
\end{array}
\]

take $T$ sufficiently positive definite and $t$ sufficiently
large; get Slater's constraint qualification!

take dual; get strengthened SDP relaxation of MC:
\[
 {\rm MCPSDP2} \begin{array}{ccc}
     \nu_2^* =
    &\max & \trace H_cY\\
  &\mbox{s.t.}& \HH^*(Y) = n\\
  &                 & \HH^*(Y) = E\\
  &                 & \HH^*(Y) = 0\\
  &                 & \HH^*(Y) = 1\\
     && Y \succeq 0.
\end{array}
\]

\end{slide}
\begin{slide}{}


We need to calculate the adjoint operators and remove redundant
constraints in MCDSDP2. Use:


\[  Y \cong \left( \begin{array}{c}y_0 \\ x \end{array}\right) 
                     \left( y_0 ~x^T\right), \quad
X = \sMat(x)
\]

simplified SDP relaxation MCPSDP2
\[
  \begin{array}{ccl}
    &\max & \trace H_cY\\
  &\mbox{s.t.}& \diag(Y) = e \\
  &                 & Y_{0,t(i)}=1, \quad \forall i=1, \ldots , n\\
  && \sum\limits_{k=1}^i Y_{t(i-1)+k,t(j-1)+k}\\
  &&  ~~+ \sum\limits_{k=i+1}^j Y_{t(k-1)+i,t(j-1)+i}\\
&&~~~~~      +\sum_{k=j+1}^nY_{t(i-1)+i,t(k-1)+j} \\
 && ~~~~~~~~~~-n Y_{0,t(j-1)+i}=0\\ 
    && ~~~~~~~~~~~~          \quad \forall 1\leq i < j \leq  n\\
     && Y \succeq 0, Y \in S^{t(n)+1}.
\end{array}
\]
This problem has $2t(n)-1$ constraints.



\end{slide}
\begin{slide}{}

surprise result

{\bf LEMMA}\\
Suppose that $Y$ is feasible in MCPSDP2. Then the first row
\[  \sMat \left( Y_{0,1:t(n)} \right) \succeq 0.  \]
\underline{\hspace{90mm}}\\
Proof:
Let $x=Y_{0,2:t(n)}$.
$X^2=nX$ constraint:
\[ n \sMat (x) = \sMat (x) \sMat (x) 
      =\sMat (x) x^T \sMat^*. 
\]
identify $xx^T$ with lower
right block of $Y$; get congruence of a positive
semidefinite matrix.
Alternatively: using ${\cal H}_3^*,$  and
$\dsvec (\cdot) \sMat$ is self-adjoint operator
\[ n\dsvec^* (x)
= \dsvec \bar{Y} \sMat 
= \left( \dsvec \bar{Y} \sMat \right)^*,
\]
where $\bar{Y}$ is the bottom
right block of $Y.$ Again congruence.
\QED


\end{slide}
\begin{slide}{}

{\bf Second Strengthened SDP Relaxation}\label{sect:2ndstrrel}

add more redundant quadratic constraints to MC2;
dual provides tighter relaxation

change of variable $X = v v^T$ gives $X_{ij} = v_iv_j$
and $v_k^2 = 1$ for $k=1,\ldots,n$; therefore
\[
X_{ij} = v_iv_j = v_i v_k^2 v_j = v_i v_k \cdot v_k v_j = X_{ik} \cdot X_{kj}
\]
interesting connection between these constraints and the metric polytope.

\end{slide}
\begin{slide}{}



Add these constraints to MC2
\[
(\mbox{MC3}) \quad   \begin{array}{ccl}
    \mu^*=
    &\max & \ \tr QX \\
  &\mbox{s.t.}& \diag(X) = e\\
     &&X\circ X=E\\
     &&X^2-nX=0\\
     &&X_{ij}=X_{ik}X_{kj}, \quad \forall\, 1 \leq i,j,k \leq n.\\
\end{array}
\]


\end{slide}
\begin{slide}{}


Taking the dual of the dual of MC3
(and removing redundant constraints in the resulting SDP)
yields the relaxation SDP3 defined below.

Alternatively, for rank-one matrices
$X = v v^T, v \in \{\pm1\}^n$.
have all their entries equal to
$\pm 1$. Hence $Y$ feasible for SDP2 have
all their entries in the first row and column equal to $\pm 1$.
Now consider the following constraints from SDP2:
\[Y_{0,T(i,j)} = \frac{1}{n} \sum\limits_{k=1}^n Y_{T(i,k),T(k,j)},
  \,\,\,\forall\,  1\leq i<j\leq n,
\]
for $Y =\pmatrix{1 & x^T \cr x & \bar{Y}}$ and $x = \svec(v v^T)$.
The entry $Y_{0,T(i,j)}$ is in the first row of $Y$ and therefore
it is equal to 1 in magnitude; and so
they must all have magnitude equal to 1.



\end{slide}
\begin{slide}{}


Either approach yields the relaxation SDP3:
\[
(\mbox{SDP3}) \quad \begin{array}{rcl}
     \nu_3^* =
    &\max & \trace H_Q Z\\
  &\mbox{s.t.}& \diag(Z) = e\\
  &    & Z_{0,t(i)}=1, i=1, \ldots , n\\
  &    & Z_{0,T(i,j)} = Z_{T(i,k),T(k,j)},
         \,\,\,\forall\, k, \forall\,1\leq i < j \leq n\\
  &    & Z \succeq 0, Z \in {\cal S}^{t(n)+1}.
\end{array}
\]


\end{slide}
\begin{slide}{}

{\bf Properties of the Second Strengthened Relaxation}

Let us define the projection of the feasible set of SDP3 onto $\Sn$ as
\[
F_n := \{ X \in \Sn : X = \sMat(Z_{1:t(n),0}), Z\,\,
\mbox{feasible for SDP3} \}.
\] 

\begin{cor}\label{IncluCor}
$C_n \subseteq F_n \subseteq {\cal E}_n \cap M_n$. \QED
\end{cor}

(cut polytope, SDP3, elliptope, metric polytope)

containment is {\em strict}
\begin{thm}
$C_n \varsubsetneq F_n \varsubsetneq {\cal E}_n \cap M_n$ for $n \geq
5$.
\QED
\end{thm}

\end{slide}

\begin{slide}{}
\begin{center} {\bf Numerical Example} \end{center}

{\large Graph:}
\begin{figure}[h]
%\centering\epsfig{file=C5.eps,height=30mm}
\centering\psfig{file=C5.eps,height=30mm}
\end{figure}

\begin{center}
{\small
\begin{tabular}{|c|c|c|c|}\hline
$\mu^*$ & SDP1 & SDP2 \\
\hline \hline
4 & 4.5225 & 4.0000 \\
  & R.E.: 13.06\% & R.E.: 0\% \\
\hline
\end{tabular}
}
\end{center}
\end{slide}

\begin{slide}{}
\begin{center} {\bf Numerical Example} \end{center}

{\large Graph:}
\begin{figure}[h]
\centering\psfig{file=AntiwebC9.eps,height=30mm}
%\centering\epsfig{file=AntiwebC9.eps,height=30mm}
\end{figure}

\begin{center}
{\small
\begin{tabular}{|c|c|c|c|}\hline
$\mu^*$ & SDP1 & SDP2 \\
\hline \hline
12 & 13.5000 & 12.4967 \\
  & R.E.: 12.50\% & R.E.: 4.14\% \\
\hline
\end{tabular}
}
\end{center}
\end{slide}

\begin{slide}{}
\begin{center} {\bf Numerical Example} \end{center}

{\large Graph:}
\begin{figure}[h]
%\centering\epsfig{file=SpecificK5.eps,height=30mm}
\centering\psfig{file=SpecificK5.eps,height=30mm}
\end{figure}

\begin{center}
{\small
\begin{tabular}{|c|c|c|c|}\hline
$\mu^*$ & SDP1 & SDP2 \\
\hline \hline
9.28 & 9.6040 & 9.2800 \\
     & R.E.: 3.49\% & R.E.: 0\% \\
\hline
\end{tabular}
}
\end{center}
\end{slide}

\begin{slide}{}
\begin{center} {\bf Numerical Example} \end{center}

{\large Graph:}
\begin{figure}[h]
%\centering\epsfig{file=K5.eps,height=30mm}
\centering\psfig{file=K5.eps,height=30mm}
\end{figure}

\begin{center}
{\small
\begin{tabular}{|c|c|c|c|}\hline
$\mu^*$ & SDP1 & SDP2 \\
\hline \hline
6 & 6.2500 & 6.2500 \\
  & R.E.: 4.17\% & R.E.: 4.17\% \\
\hline
\end{tabular}
}
\end{center}


\end{slide}
\begin{slide}{}

\begin{center} {\bf Ising spin glass problems} \end{center}

\vspace{10mm}

All problems have short range interactions only,

periodic boundary conditions

and were generated using the {\tt torusgen} codes from J\"unger et al.

\end{slide}
\begin{slide}{}

\begin{figure}[h]
\centering\psfig{file=torus50.ps,height=60mm}
\end{figure}
\end{slide}

\begin{slide}{}
\begin{center}
{\tiny
\begin{table}[htbp]
\begin{tabular}{|c|c|c|c|c|c|c|}\hline
Problem & Number of & SDP1   & SDP2   & SDP1     & SDP2 \\
type    & particles & $\frac{\mu^*}{\nu_1^*}$ & $\frac{\mu^*}{\nu_2^*}$ & found rank-1   & found rank-1 \\
        &           &        &        & optimum  & optimum\\
\hline
\hline
         & 12 & 0.9825 &  1.0000 & No & Yes \\
\cline{2-6}
 2-D     & 20 & 0.9775 &  1.0000 & No & Yes \\
\cline{2-6}
Gaussian & 20 & 0.9666 &  1.0000 & No & Yes \\
\cline{2-6}
         & 24 & 0.9288 &  1.0000 & No & Yes \\
\cline{2-6}
         & 16 & 0.9985 &  1.0000 & No & Yes \\
\hline
         & 16 & 0.9999 &  1.0000 & No & Yes \\
\cline{2-6}
 3-D     & 18 & 0.9720 &  1.0000 & No & Yes \\
\cline{2-6}
Gaussian & 12 & 0.9730 &  1.0000 & No & Yes \\
\cline{2-6}
         & 16 & 0.9192 &  1.0000 & No & Yes \\
\cline{2-6}
         & 20 & 0.9942 &  1.0000 & No & Yes \\
\hline
         & 12 & 0.8323 &  1.0000 & No &  No \\
\cline{2-6}
 2-D     & 16 & 0.9205 &  1.0000 & No &  No \\
\cline{2-6}
$\pm 1$  & 16 & 0.7379 &  1.0000 & No &  No \\
\cline{2-6}
         & 16 & 0.9859 &  1.0000 & No & Yes \\
\cline{2-6}
         & 20 & 0.8863 &  1.0000 & No &  No \\
\hline
         & 12 & 0.9951 &  1.0000 & No & Yes \\
\cline{2-6}
         & 18 & 0.9470 &  1.0000 & No & Yes \\
\cline{2-6}
 3-D     & 18 & 0.8711 &  1.0000 & No &  No \\
\cline{2-6}
$\pm 1$  & 18 & 0.9075 &  1.0000 & No &  No \\
\cline{2-6}
         & 18 & 0.9642 &  1.0000 & No & Yes \\
\cline{2-6}
         & 18 & 0.8428 &  1.0000 & No & Yes \\
\hline
\end{tabular}
\end{table}
}
\end{center}
\end{slide}

\begin{slide}{}
{\bf Observations:}
\begin{enumerate}
\item SDP2 {\em always} found the weight of the optimal cut, for
all problems.
\item Furthermore, SDP2 seems to often find an optimal rank-one matrix
and hence an optimal cut. In fact,
SDP2 found an optimal cut for all 10 problems with
Gaussian distributed couplings, and also for almost half
(5 out of 11) of the
problems with $\pm$ 1 couplings.
\item {\em Why?}
Perhaps looking into the structure of the specific 
test problems will help us to understand this phenomenon.
\end{enumerate}

\end{slide}
\begin{slide}{}




\oddsidemargin -.35in
\evensidemargin -.35in
{\tiny

%\begin{table}[b]
%\begin{center}
\begin{tabular}{||c|c|c|c|c|c|c|c|c||}\hline
 $n$ & Wt & MCSDP& MCPSDP2&
 Num. \\
     & opt.& (\% rel. err) &
 (\% rel. err) & rank \\
\hline \hline
5 & 4 & 4.5225 (13.06\%) & 4.2890 (7.22\%) & 2 \\
\hline
7 & 56 & 56.4055 (0.72\%) & 56.0954 (0.17\%) & 3 \\
\hline
8 & 30 & 30.2015 (0.67\%) & 30. (e-10\%) & 1 \\
\hline
9 & 58 & 58.9361 (1.61\%) & 58.1182 (0.20\%) & 3 \\
\hline
 10&     64&  64.08 (0.1268\%) &   64 (e-08\%) &  3 \\
\hline
12 & 88 & 90.3919 (2.72\%) & 89.5733 (1.79\%) & 4 \\
\hline
\end{tabular}
%\end{center}
%\end{table}


The first line of results corresponds to
solving both MC relaxations for a 5-cycle with unit edge-weights;
the others come from randomly generated weighted graphs.

}




\end{slide}
\begin{slide}{}

\oddsidemargin -.75in
\evensidemargin -.75in
{\tiny

%\begin{table}[b]
%\begin{center}
\begin{tabular}{|c|c|c|c|c|c|c|}\hline
Graph & $\mu^*$ & SDP1  & $\rm{SDP2}_{\rm{P}}$ & $M_n$ &
        ${\cal E}_n \cap M_n$ & $\rm{SDP3}_{\rm{P}}$\\
      &         & bound & bound & bound & bound                 & bound \\
 & & & & & & \\
\hline \hline
$C_5$ & 4 & 4.5225 & 4.2889 & 4.0000 & 4.0000 & 4.0000 \\
  &   & $\rho$ = 0.8845 & $\rho$ = 0.9326 & $\rho$ = 1.0000 &
  $\rho$ = 1.0000 & $\rho$ = 1.0000 \\
  &   & R.E.: 13.06\% & R.E.: 7.22\% & R.E.: 0\% & R.E.: 0\% & R.E.: 0\% \\
\hline
$K_5 \backslash e$ & 6 & 6.2500 & 6.1160 & 6.0000 & 6.0000 & 6.0000 \\
  &   & $\rho$ = 0.9600 & $\rho$ = 0.9810 & $\rho$ = 1.0000 &
  $\rho$ = 1.0000 & $\rho$ = 1.0000 \\
  &   & R.E.: 4.17\% & R.E.: 1.93\% & R.E.: 0\% & R.E.: 0\% 
  & R.E.: 0\% \\
\hline
$K_5$ & 6 & 6.2500 & 6.2500 & 6.6667 & 6.2500 & 6.2500 \\
  &   & $\rho$ = 0.9600 & $\rho$ = 0.9600 & $\rho$ = 0.9000 &
  $\rho$ = 0.9600 & $\rho$ = 0.9600 \\
  &   & R.E.: 4.17\% & R.E.: 4.17\% & R.E.: 11.11\% & R.E.: 4.17\%
  & R.E.: 4.17\% \\
\hline
Given      & 9.28 & 9.6040 & 9.4056 & 9.3867 & 9.2961 & 9.2800 \\
by $A(G)$  & & $\rho$ = 0.9663 & $\rho$ = 0.9866 & $\rho$ = 0.9886 &
  $\rho$ = 0.9983 & $\rho$ = 1.0000 \\
($n=5$) & & R.E.: 3.49\% & R.E.: 1.35\% & R.E.: 1.15\% 
 & R.E.: 0.17\% & R.E.: 0\% \\
\hline
 & 12 & 13.5 & 12.9827 & 12.8571 & 12.6114 & 12.4967 \\
${\rm AW}^2_9$ & & $\rho$ = 0.8889 & $\rho$ = 0.9243 & $\rho$ = 0.9333 &
  $\rho$ = 0.9515 & $\rho$ = 0.9603 \\
& & R.E.: 12.50\% & R.E.: 8.19\% & R.E.: 7.14\% & R.E.: 5.10\% &
  R.E.: 4.14\% \\
\hline
         & 12 & 12.5  & 12.3781 &12.0000 & 12.0000 & 12.0000 \\
Pet.     &   & $\rho$ = 0.9600 & $\rho$ = 0.9695 & $\rho$ = 1.0000 &
  $\rho$ = 1.0000 & $\rho$ = 1.0000 \\
($n=10$) &   & R.E.: 4.17\% & R.E.: 3.15\% &R.E.: 0\% & R.E.: 0\% & R.E.: 0\% \\
\hline
Given in  & 88 & 90.3919 & 89.5733 & 89.3333 & 88.0029 & 88.0000 \\
\cite{AnWo:00}& & $\rho$ = 0.9735 & $\rho$ = 0.9824 & $\rho$ = 0.9851 &
  $\rho$ = 1.0000 & $\rho$ = 1.0000 \\
($n = 12$)  & & R.E.: 2.72\% & R.E.: 1.79\% & R.E.: 1.52\% & R.E.: $3.3E-5$ &
  R.E.: $9.9E-7$ \\
\hline
\end{tabular}
%\end{center}
%\end{table}
{Numerical comparison of all MC relaxations for selected test problems}
}


\end{slide}
\begin{slide}{}
\oddsidemargin -.15in
\evensidemargin -.75in
{\tiny
%\begin{table}[b]
%\begin{center}
\begin{tabular}{|c|c|c|c|c|}\hline
Number   & $\mu^*$ & SDP1  & $\rm{SDP2}_{\rm{P}}$ & $\rm{SDP3}_{\rm{P}}$ \\
of       &         & bound & bound                & bound \\
vertices & & & & \\
\hline \hline
 10 & 648 & 666.428 & 656.8020 & 648.000 \\
    &   & $\rho$ = 0.9723 & $\rho$ = 0.9866 & $\rho$ = 1.0000 \\
    &   & R.E.: 2.84\% & R.E.: 1.36\% & R.E.: 0\% \\
\hline
 11 & 1060 & 1084.345 & 1072.352 & 1060.000 \\
    &   & $\rho$ = 0.9775 & $\rho$ = 0.9885 & $\rho$ = 1.0000 \\
    &   & R.E.: 2.30\% & R.E.: 1.17\% & R.E.: 0\% \\
\hline
 15 & 2290 & 2317.354 & 2301.634 & 2290.000 \\
    &   & $\rho$ = 0.9882 & $\rho$ = 0.9949 & $\rho$ = 1.0000 \\
    &   & R.E.: 1.19\% & R.E.: 0.51\% & R.E.: 0\% \\ 
\hline
 16 & 2270 & 2318.867 & 2300.354 & 2270.000 \\
    &   & $\rho$ = 0.9789  & $\rho$ = 0.9868 & $\rho$ = 1.0000 \\
    &   & R.E.: 2.15\% & R.E.: 1.34\% & R.E.: 0\% \\
\hline
 25 & 380 & 385.4737 & 383.6503 & 380.000 \\
    &   & $\rho$ = 0.9858 & $\rho$ = 0.9905 & $\rho$ = 1.0000 \\
    &   & R.E.: 1.44\% & R.E.: 0.96\% & R.E.: 0\% \\
\hline
 30 & 1705.5 & 1751.600 & 1743.205 & 1705.578 \\
    &   & $\rho$ = 0.9737 & $\rho$ = 0.9784 & $\rho$ = 1.0000 \\
    &   & R.E.: 2.70\% & R.E.: 2.21\% & R.E.: $4.6E-5$ \\
\hline
 33 & 1888.5 & 1932.968 & 1926.119 & 1888.564 \\
    &   & $\rho$ = 0.9770 & $\rho$ = 0.9805 & $\rho$ = 1.0000 \\
    &   & R.E.: 2.35\% & R.E.: 1.99\% & R.E.: $3.4E-5$ \\
\hline
 36 & 27108.55 & 28305.28 & 27944.30 & 27108.81 \\
    &   & $\rho$ = 0.9577 & $\rho$ = 0.9701 & $\rho$ = 1.0000 \\
    &   & R.E.: 4.41\% & R.E.: 3.08\% & R.E.: $9.8E-6$\ \\
\hline
\end{tabular}
%\end{center}
%\end{table}
~\\
{Numerical comparison of SDP1, $\rm{SDP2}_{\rm{P}}$ and
$\rm{SDP3}_{\rm{P}}$ on randomly generated graphs with
non-negative edge weights}
}

\label{endsect:appl}
\end{slide}
\begin{slide}{}

\subsection{Quadratically Constrained Quadratic Programs and SDP
Relaxations}
\label{sect:QQPSDP}
{\bf Outline}

\begin{enumerate}
\item[$\bullet$]
Theme: Lagrangian Relaxation is Best 
\item[$\bullet$]
Examples of QQPs with Zero Duality Gaps
\item[$\bullet$]
Advantages of Redundant Constraints
(more in Section \ref{sect:QBP} on Quadratic Boolean Programming)
\item[$\bullet$]
Applications to QAP, GP

\end{enumerate}




\end{slide}
\begin{slide}{}

\subsubsection{Quadratically Constrained Programs with Zero Duality Gaps}
\label{sect:orthogconstr}


\[\mbox{Let} \quad q_i(y)=\frac 12 y^TQ_iy+y^Tb_i + c_i,~y\in \Rn \]
Consider the NP-hard
\[ {\bf (QQP)} \qquad
\left\{ \begin{array}{ccc}
    q^*=  & \min &q_0(y) \\
 &  \mbox{s.t.} & q_i(y) \leq 0\\
     &&  i=1,\ldots m
    \end{array}   \right.
\]
$\bullet$ Lagrangian is
$ L(y,\lambda) = q_0(y) + \sum_{i=1}^m \lambda_i q_i(y) $
\[\mbox{$\bullet$ Primal-Dual pair:} \quad 
    q^*=\min_y \max_{\lambda \geq 0} L(y,\lambda) \geq d^* 
                := \max_{\lambda \geq 0} \min_y L(y,\lambda)
\]


\end{slide}
\begin{slide}{}

Using the hidden constraint (a quadratic is bounded below only if the
Hessian is positive semidefinite), we get:\\
the Lagrangian dual is equivalent to an SDP - thus tractable!

When do duality gaps exist? Zero duality gaps exist?

\begin{quote}
Theme: Lagrangian duality is {\em best} relaxation.
(Tractable relaxations can be obtained
using Lagrangian relaxation.)
\end{quote}



\end{slide}
\begin{slide}{}


Example: {\bf Rayleigh Quotient, $A=A^T$}\\
(Finding the eigenvalues is a tractable problem - well known.)

smallest eigenvalue $\lambda_1$ of $A$
\beq \label{eq:rayleigh}
 \lambda_1 = \min \{x^T Ax : x^T  x = 1\}.
\eeq
$A$ is not necessarily positive semidefinite, so this is the
minimization of a nonconvex function on a nonconvex set. 
There is no duality gap for this nonconvex problem, i.e.
\beq \label{eq:rayleighdual}
 \lambda_1 = \max_{\lambda}\ \min_x\ x^T Ax -\lambda (x^T  x - 1)
 = \max_{A-\lambda I \succeq 0}\ \lambda.
\eeq
use the hidden semidefinite constraint 
\[  
A - \lambda I \succeq 0,
\]

\end{slide}
\begin{slide}{}


Example: {\bf Trust Region Subproblem}
\begin{eqnarray*}
 &\mu^*:=\min& q_0(x)\\
&\mbox{s.t.}& x^T x - \delta^2\leq 0 \mbox{ (or } =0). 
\end{eqnarray*}
This is a tractable problem, e.g. \cite{Gay:81,MoSo:83}

(allow more general nonconvex constraints $\alpha \leq q(x) \leq \beta.$)
for ``$\le$," the Lagrangian dual is:
\[ 
\mbox{DTRS}\qquad
 \nu^*:=\max\limits_{\lambda \geq 0}\ \min\limits_x\ q_0(x) + \lambda 
(x^T x - \delta^2).
\]
strong duality holds (\cite{sw5}), a dual is the
(concave) nonlinear semidefinite program 
\begin{eqnarray*}
\mbox{DTRS}\qquad
&\nu^*:=\max& g_0^T  (Q+\lambda I)^{\dagger} g_0 - \lambda \delta^2\\
&\mbox{s.t.}& Q+\lambda I \succeq 0\\
&&\lambda \geq  0.
\end{eqnarray*}



\end{slide}
\begin{slide}{}

short proof of strong duality (e.g. \cite{Lewco663:94})\\
WLOG TRS is nonconvex; 
smallest eigenvalue of $Q_0,$ denoted $\gamma,$ is negative:
\[
\begin{array}{rccll}
 && \mu^* = \\
& =& \min\limits_{x^Tx \leq \delta^2}
       & x^T(Q_0-\gamma I)x-2c_0^tx + \gamma x^Tx\\
 &= & \min\limits_{x^Tx = \delta^2}
       & x^T(Q_0-\gamma I)x-2c_0^tx + \gamma x^tx, 
                     \quad \mbox{($Q_0$ is indefinite)}\\
 &= & \min\limits_{x^Tx = \delta^2}
       & x^T(Q_0-\gamma I)x-2c_0^tx + \gamma \delta^2\\
 &= & \min\limits_{x^Tx \leq \delta^2}
       & x^T(Q_0-\gamma I)x-2c_0^tx + \gamma \delta^2,
            \mbox{ ($Q_0-\gamma I$ singular)}\\
 &=&   \max\limits_{\lambda \geq 0} \min\limits_x
     & x^T(Q_0-\gamma I)x-2c_0^tx + \lambda(x^Tx-\delta^2) + \gamma \delta^2,
     \mbox{ (convex case)}\\
 &=&   \max\limits_{\lambda \geq 0} \min\limits_x
       & x^TQ_0x-2c_0^tx +(\lambda - \gamma) (x^Tx-\delta^2)\\
 &\leq &   \max\limits_{\lambda \geq \gamma} \min\limits_x
       & x^TQ_0x-2c_0^tx +(\lambda - \gamma) (x^Tx-\delta^2)
           \quad \mbox{($\gamma < 0$)}\\
 &=& \nu^* \leq   \mu^*.
\end{array}
\]



\end{slide}
\begin{slide}{}

{\bf  Two Trust Region Subproblem}
TTRS consists in minimizing a
(possibly nonconvex) quadratic function subject to a norm and a
least squares constraint.

ref in SQP methods by Celis-Dennis-Tapia.

TTRS can have a nonzero duality gap, ref Peng-Yuan.

if objective not convex,
then the primal may not be attained, ref Luo-Zhang.

TRS can have at most one local and nonglobal optimum, ref Martinez.

Still an open problem whether TTRS is an NP-hard or a polynomial time
problem.


\end{slide}
\begin{slide}{}

Note: a quadratic constraint can be written as
\[  y^tPy \leq \delta  \]
after the homogenization. This is lifted to
\[  \trace PY \leq \delta. \]
Second lifting:
\[   YPY \preceq \delta Y. \]
This strictly strengthens the SDP relaxation.



\end{slide}
\begin{slide}{}


{\bf Orthogonal Contraints}\\
permutation matrices are a subset of orthogonal matrices
\[ \Pi \subset {\cal O} := \{X: X X^T =I\} \]
(Stiefel manifold ref e.g. Edelman,Arias,Smith)

$A$ and $B$ $n\times n$ symmetric matrices
\[
\begin{array}{rcl}
{\rm QQP_O}\qquad  \mu^*:=&\min& \tr AXBX^T\\
&{\rm s.t.}& XX^T=I.
\end{array}
\]
({\em looks} like TRS)


\end{slide}
\begin{slide}{}

${\rm QQP_O}$ is tractable; solved using e.g.
classical Hoffman-Wielandt inequality; provides bounds for e.g. QAP.

\bpr Apply Lagrange multipliers, $S=S^T$.  Lagrangian is:
\begin{eqnarray*}
L(X,S) &=& \tr (AXBX^T) + \left< S,(XX^T-I) \right>\\
 &=& \tr (AXBX^T) + \trace S(XX^T-I)
\end{eqnarray*}
\[ 0= \left< \nabla L(X,S),h \right> = 2 \trace AXB h^T + SXIh^T
  \quad \forall h.
\]
Therefore $AXBX^T = -S=-S^T$, i.e. $A$ and $XBX^T$ are mutually
diagonalizable. The optimal value is then 
\beq 
\label{minimal_product}
\mu^* = \sum_i \lambda_i(A) \lambda_{n-i+1}(B)
\eeq
(minimal scalar product of eigenvalues)
\epr

\end{slide}
\begin{slide}{}


But, Lagrangian dual can have a duality gap, \cite{KaReWoZh:94}

Lagrangian of $(P)$
$$
L(X,S) = \tr AXBX^T + \tr SXX^T - \tr S.
$$
Lagrangian dual is
\[
\nu^* = \max_S \min\limits_X  L(X,S).
\]
The hidden constraint (Hessian is psd) yields the dual
\[
(D)
\begin{array}{ccc}
\mu^D = &\max\limits_{\hat{S}=\hat{S}^T} & -\tr \hat{S} \\ 
&  \mbox{~subject to~}  &  (B \otimes A + I \otimes \hat{S}) \succeq 0. 
\end{array}
\]


\end{slide}
\begin{slide}{}

\begin{exam}
$$
A:= \left( 
\begin{array}{cc}
1 & 0 \\
0 & 2 
\end{array} 
\right) ~~~~~~~~
B:= \left( 
\begin{array}{cc}
3 & 0 \\
0 & 4 
\end{array} 
\right).
$$
$\mu^* = 10$
$$
B \otimes A = \left( 
\begin{array}{cccc}
3 & 0 & 0 & 0 \\
0 & 6 & 0 & 0 \\
0 & 0 & 4 & 0 \\
0 & 0 & 0 & 8 \\
\end{array} 
\right).
$$
But $s_{11}\geq -3$ and $s_{22}\geq -6$;
to maximize $(D)$, equality must hold, and therefore $-\tr \hat{S} = 
9$ in the optimum. 

However, add redundant constraint $X^TX=I$.
Get $T \otimes I$ in Hessian and $-\trace T$ in objective function.
$s_{11}\geq -3-t_{11}$ 
$s_{11}\geq -4-t_{22}$ 
$s_{22}\geq -6-t_{11}$
$s_{22}\geq -8-t_{22}$
So $t_{22}=-1.$ Duality gap is closed.

\end{exam}


\end{slide}
\begin{slide}{}


Summary: constraints $XX^T=I$ and
$X^T X=I$ are equivalent. Add redundant constraints.
\begin{eqnarray*}
{\rm QQP_{OO}}\qquad  \mu^O:=&\min& \tr AXBX^T\\
&{\rm s.t.}& XX^T=I,\ X^T X=I.
\end{eqnarray*}
dual problem is
\begin{eqnarray*}
{\rm DQQP_{OO}}\qquad
\mu^O \geq \mu^D:=&\max& \tr S+\tr T\\
&\mbox{s.t.}&
(I\otimes S)+(T\otimes I)\preceq (B\otimes A)\\
&& S=S^T,\ T=T^T.
\end{eqnarray*}

\begin{thm} (\cite{AnWo:98})
\label{strong_duality}
Strong duality holds for $\rm QQP_{OO}$ and $\rm DQQP_{OO},$ 
i.e.  $\mu^D=\mu^O$ and both primal and dual are attained.
\end{thm}

Theme again holds.

\end{slide}
\begin{slide}{}



\bpr
(of Theorem \ref{strong_duality})
Let $A=V\Sigma V^T$, $B=U\Lambda U^T$, where $V$ and $U$ are orthonormal 
matrices whose columns are the eigenvectors of $A$ and $B$, respectively,
$\sigma$ and $\lambda$ are the corresponding vectors of eigenvalues, and
$\Sigma=\Diag(\sigma)$, $\Lambda=\Diag(\lambda)$. Then for any $S$ and $T$,
\[
\begin{array}{l}
(B\otimes  A)-(I\otimes  S)-(T\otimes  I)=\\
\qquad \qquad 
(U\otimes  V)\left[(\Lambda\otimes  
       \Sigma)-(I\otimes  \bar{S})-(\bar{T}\otimes  I)\right]
                 (U^T\otimes  V^T),
\end{array}
\]
where $\bar{S}=V^T SV$, $\bar{T}=U^T T U$. Since $U\otimes  V$ is nonsingular,
$\tr S=\tr \bar{S}$ and $\tr T=\tr \bar{T}$, 
the dual problem ${\rm DQQP_{OO}}$ is equivalent to
\[
\begin{array}{rcl}
\mu^D=&\max& \tr S+\tr T\nonumber\\
&\mbox{s.t.}&(\Lambda\otimes  \Sigma)-(I\otimes  S)-(T\otimes  I)\succeq 0
\label{dual_problem2}\\
&&S=S^T,\ T=T^T.\nonumber
\end{array}
\]



\end{slide}
\begin{slide}{}


However, since $\Lambda$ and $\Sigma$ are diagonal matrices, 
\req{dual_problem2} is equivalent to the ordinary linear program:
\begin{eqnarray*}
{\rm LD}\qquad&\max&  e^T s+e^T t\\
&\mbox{s.t.}&\lambda_i\sigma_j-s_j-t_i\ge 0,\quad i,j=1,\ldots, n.
\end{eqnarray*}
But LD is the dual of the linear assignment problem:
\begin{eqnarray*}
{\rm LP}\qquad&\min& \sum_{i,j} \lambda_i\sigma_j x_{ij} \\
&\mbox{s.t.}&\sum_{j=1}^n x_{ij} = 1,\quad i=1,\ldots,n\\
&&\sum_{i=1}^n x_{ij} = 1,\quad j=1,\ldots,n\\
&&x_{ij} \ge  0,\quad i,j=1,\ldots,n.
\end{eqnarray*}

\end{slide}
\begin{slide}{}

Assume without loss of generality that $\lambda_1\le\lambda_2\le\ldots\le\lambda_n$,
and $\sigma_1\ge\sigma_2\ge\ldots\ge\sigma_n$. Then LP can be interpreted
as the problem of finding a permutation $\pi(\cdot)$ of $\{1,\ldots,n\}$ 
so that $\sum_{i=1}^n \lambda_i\sigma_{\pi(i)}$ is minimized. 
But the minimizing 
permutation is then $\pi(i)=i$, $i=1,\ldots,n$, and 
from the minimal inner-product expression
\req{minimal_product}, the solution value
$\mu^D$ is exactly $\mu^O$.\qquad
\epr


\end{slide}
\begin{slide}{}


Other applications:

Weighted Sums of Eigenvalues;

Graph Partitioning Problem;

TRS like constraints $\{X: XX^T \preceq I\}$ (\cite{AnChWoYu:98})
(extension of Hoffman-Wielandt inequality)



\label{endsect:orthogconstr}

\end{slide}
\begin{slide}{}


\subsubsection{Quadratic Assignment Problem, QAP}
\label{sect:QAP}
\[
(QAP)~~ \begin{array}{ccc}
\mu^*:= &\min\limits_{X \in \Pi} & \tr AXBX^T - 2CX^T, \\
\end{array}
\]
$A, B$ real symmetric $n\times n$ matrices, $C$ 
real $n\times n$ matrix, $\Pi$ 
set of permutation matrices.

model for e.g.
allocating set of $n$ facilities to set of $n$ locations while
minimizing quadratic objective (distance-flow)

QAP is $N\!P$-hard and, in practice, 
problems of 
moderate sizes, such as $n=16$, are still considered very hard. 
(Recent advances to $n=30$, \cite{AnsBri:00}.)


\end{slide}
\begin{slide}{}


Questions for SDP type bounds:
\begin{enumerate}
\item
interesting numerical and theoretical difficulties, e.g. loss of
constraint qualification and  loss of sparsity in the optimality
conditions.
\item
Can the new bound compete with other bounding techniques in speed and
quality?
\item
comparison on
the Nugent test problems?
%\item
%Can we incorporate the new bound in a branch and bound algorithm?
\item
add new facet inequalities?
\end{enumerate}

\end{slide}
\begin{slide}{}


Notation:\\

${\mathcal E} := \{ X : Xe=X^Te=e \}$\\
set of matrices satisfying {\em assignment constraints}

${\mathcal Z} := \left\{ X : X_{ij} \in \{0,1\} \right\}$ \\
set of (0,1)-matrices

${\mathcal N} := \left\{ X : X_{ij} \geq 0 \right\}$  \\
set of {\em nonnegative matrices}

${\mathcal O} := \{ X : XX^T=X^TX=I \}$\\
set of orthogonal matrices

\end{slide}
\begin{slide}{}


\[
 \Pi = {\mathcal E} \cap {\mathcal Z} = {\mathcal O} \cap {\mathcal Z},
\qquad \Diag \leftrightarrow  \diag \mbox{ is vector }
\leftrightarrow \mbox{ matrix }
\]
rewrite QAP as (add redundant constraints)
\[
\begin{array}{ccl}
&\min & \tr AXBX^T - 2CX^T \\
&\mbox{~s.t.~} & XX^T =  X^TX = I \\
              && ||Xe -  e||^2=0 \\
              && || X^Te - e||^2=0 \\
        && X_{ij}^2 - X_{ij} = 0, \quad \forall i,j.\\
        &&  \diag(X_{:i}X_{:j}^T)=0, \quad \mbox{if } i \neq j\\
        &&  X_{:i}X_{:j}^T - \Diag\left(\diag(X_{:i}X_{:j}^T)\right) 
                    =0, \quad \mbox{if } i = j\\
\end{array}
\] 

\end{slide}
\begin{slide}{}


{Lagrangian Relaxation}\\
\[
\begin{array}{ll}
\mu_{\mathcal O} \geq  \mu_{\mathcal L} \\
~~~:= \max\limits_{W,u_{0},v_{0}\ldots} \min\limits_{XX^T=X^TX = I}
&\{\tr AXBX^T - 2CX^T\\
&~~+ \sum_{ij} W_{ij}(X_{ij}^2 - X_{ij})\\
&~~~+u_{0}\|Xe-e\|^{2}\\
&~~~~+v_0\|X^Te-e\|^{2}\\
&~~~~+\ldots \}.
\end{array}
\]
homogenize the Lagrangian using
scalar $x_0$ and constraint $x_{0}^{2}=1.$


\end{slide}
\begin{slide}{}

We get the lower bound\\
 (quadratic, linear constant in $X$)
\[
\begin{array}{ll}
         \max\limits_{W,S_b,S_o,u_{0}v_0,,w_{0}}  \min\limits_{X,~x_0} 
                     & \{ \tr [ AXBX^T \\
& ~~+ u_{0}\|Xe\|^2+v_0 \|X^Te\|^2 \\
& ~~~+ W(X \circ X)^T + w_0 x_0^2 \\
& ~~~~~+ S_b XX^T + S_o X^TX ] \\
& ~- \tr x_0(2C+ W)X^T\\
&~~~~ - 2x_{0}u_{0}e^T(X+X^T)e \\
&~~~~~ + \ldots \\
& ~- w_0 - \tr S_b - \tr S_o +2nu_{0}x_0^2 \}.
\end{array}
\]



\end{slide}
\begin{slide}{}

The hidden semidefinite constraint yields the equivalent SDP:
\[
(D_{\cal O})~~
\begin{array}{llc}
\max & - w_0 - \tr S_b - \tr S_o \\
\mbox{~s.t.~}& L_Q +\Arrow(w) + \\
        & \BoDiag(S_b) + \OoDiag(S_o) + w_0D \succeq 0,
    \end{array}
\]
The dual of this dual yields the semidefinite relaxation.

$Y \succeq 0$ is $(n^2+1) \times ( n^2+1)$\\
     the dual matrix variable

\[
(SDP_{\cal O})~~
\begin{array}{cllcll}
\min &\tr L_QY \\
\mbox{~s.t.~}
  & \bodiag(Y) = I && \oodiag(Y) = I \\
  & \arrow(Y) = e_{0} &&  \trace DY=0 \\
  & Y \succeq 0
\end{array}
\]
\end{slide}
\begin{slide}{}
adjoint operators are:
\[\arrow (Y) := \diag (Y) - (0, (Y_{0,1:n^2})^t. \]

$$ \bodiag(Y) := \sum\limits_{k=1}^n Y_{(k-1)n+1:kn,(k-1)n+1:kn} $$

$$ [\oodiag(Y)]_{ij} := \tr Y_{(i-1)n+1:in,(j-1)n+1:jn}  $$



\end{slide}
\begin{slide}{}

check Slater's condition in SDP relaxation

\[
\begin{array}{cll}
\min &\tr L_QY \\
\mbox{~s.t.~}
  & \bodiag(Y) = I, & \oodiag(Y) = I \\
  & \arrow(Y) = e_{0}, & \tr DY=0 \\
  & \ldots \\
  & Y \succeq 0,
\end{array}
\]

 $D \succeq 0$ so it fails!!\\
But we can project onto the minimal face!!

Now remove redundant constraints to get simplified SDP relaxation


\end{slide}
\begin{slide}{}

simple projected relaxation with
$n^{3}-2n^{2}+1$ constraints.

\[
\begin{array}{lll}
\mu_{R2} := & \min & \tr (\hat{V}^TL_Q\hat{V})R   \\
 & \mbox{~s.t.}
 & {\mathcal G}_{\bar{J}}(\hat{V}R\hat{V}^T) = E_{00}  \\
 &&  R \succeq 0.  
\end{array}
\]

The dual problem is
\[
\begin{array}{lll}
\mu_{R2} = & \max & -Y_{00} \\
& \mbox{~s.t.}
&\hat{V}^{T}(L_{Q}+{\mathcal G}^{*}_{\bar{J}}(Y))\hat{V} \succeq 0.
\end{array}
\]

\end{slide}
\begin{slide}{}



\begin{center}
{\bf Direct Approach to SDP Relaxation}
\end{center}

Let\\
 $X \in \Pi_n$ be a permutation matrix\\
$x=\kvec(X),~ c=\kvec(C).$
\begin{eqnarray*}
  q(X) &=& \tr AXBX^t - 2CX^t \\
       &=& x^t (B \otimes A) x -2c^tx \\
       &=& \trace xx^t (B \otimes A)  -2c^tx \\
       &=& \trace L_Q Y_X,
\end{eqnarray*}
where $L_Q$ is as above and
\[
Y_X := \left[
\begin{array}{cc}
1& x^t \\
x & xx^t
\end{array}
\right].
\]
\end{slide}
\begin{slide}{}
\begin{center}
{\bf Adding Generic Inequality Constraints}
\end{center}

From the relaxation for the (0,1)-constraints of the original problem:
$$
y_{ij} \geq 0 \mbox{~~since~~} x_i x_j \geq 0.
$$
We also get so called triangle inequalities 
\[
y_{ij}+y_{ik}+y_{jk}-\left( y_{ii}+y_{jj}+y_{kk}\right)+1 \geq 0
\]
and
\[
y_{ij}-y_{ik}-y_{jk}+y_{kk} \geq 0
\]

\end{slide}
\begin{slide}{}

{\bf THEOREM}
Let $x=\kvec(X)$ and $F_S$ be the feasible set of (SDP).
Define the centroid point
\[
\hat{Y} := \frac 1{n!} \sum\limits_{X \in \Pi_n}
 \left[
\begin{array}{cc}
1& x^t \\
x & xx^t
\end{array}
\right].
\]
Then:\\

1.~~
$\hat{Y}$ has a 1 in the (1,1) position and $n$ diagonal $n \times n$
blocks with diagonal elements $1/n.$ The first row and column equal
the diagonal.  The rest of the matrix is made up
of  $n \times n$ blocks with all elements equal to $1/(n(n-1))$  except
for the diagonal elements which are 0:
\end{slide}
\begin{slide}{}
\[
{\tiny
\begin{array}{ll}
\hat{Y} =\\
{   %\tiny
=\left[
\begin{array}{c|c}
1 & \frac 1n e^t \\ \hline\\
\frac 1n e & \begin{array}{cccc}
             \diag(\frac 1n e) &  (\frac 1{n(n-1)}) (E-I)
       &\cdots &(\frac 1{n(n-1)}) (E-I) \\
                      \cdots  &\cdots  &\cdots  & \cdots  \\
                      \cdots  &\cdots  &\cdots  & \cdots  \\
                      \cdots  &\cdots  &\cdots  & \cdots  \\
                    (\frac 1{n(n-1)}) (E-I) &\cdots
   &(\frac 1{n(n-1)}) (E-I) & \diag(\frac 1n e)
             \end{array}
\end{array}
\right] } \\
~\\
=\left[
\begin{array}{c|c}
1 & \frac 1n e^t \\
\hline\\
\frac 1n e &
E \otimes \left( \frac 1{n(n-1)} (E-I)\right)
       - I \otimes \left( \frac 1{n(n-1)} E - \frac
            1{n-1} I\right)
\end{array}
\right]
\end{array}  }
\]
2.~~
\[
\rank(\hat{Y}) = (n-1)^2 +1
\]



\end{slide}
\begin{slide}{}

3.~~
\[
\diag \hat{Y} =
 \left(
\begin{array}{c}
1 \\
\frac 1n e
\end{array}
\right)
\]
4.~~
\[
\trace(\hat{Y}) = 1 +n
\]


5.~~\\
The $n^2+1$ eigenvalues of $\hat{Y}$ are given in the vector
\[
(2,\frac 1{n-1} e_{(n-1)^2},0 e_{2n-1})
\]

\end{slide}
\begin{slide}{}

6.~~
\[
{\cal N} (\hat{Y}) = \left\{
\left( \begin{array}{c}   -\frac 1n e^tu\\ u
          \end{array}  \right)
          : u \in {\cal R} (T^t) \right\},
\]
where $T$ is the assignment constraint matrix.

-------------------------------

We can use the matrix $T$ to project onto the minimal face.

We get a simplified SDP with a positive definite feasible point.


\epr



\end{slide}
\begin{slide}{}

Recent results of Anstreicher uses a weakened form of the SDP relaxation
for QAP in combination with the strong duality result for the eigenvalue
bound.


\label{endsect:QAP}

\end{slide}
\begin{slide}{}


\subsubsection{Graph Partitioning, GP}
\label{sect:GP}

\begin{center}
(special case of QAP)
\end{center}
$k$ and $m$  given integers\\
 $G$ an edgeweighted undirected graph on $n :=km$ nodes, 
given by its adjacency matrix $A$.\\
($a_{ij}$ weight of edge $i  \leftrightarrow j$)

a $k$-partition of $V(G)$ is a partitioning of $V(G)$ into $k$ subsets
$(S_1, \ldots, S_k)$ of equal cardinality.

\begin{center}
\mbox{denoted}~~~(k-GP)
\end{center}

the columns of $Y \in \Re^{n \times k}$ are the characteristic vectors
for the sets $S_k$
$${\cal F}_k := \lbrace Y: Yu_k = u_n, 
   ~Y^tu_n= mu_k, y_{ij} \in \lbrace 0,1
\rbrace \rbrace$$

$L := \diag(Au_n) - A$ ~~{\em Laplacian matrix associated to $G$}.

\end{slide}
\begin{slide}{}
weight of the edges of $G$, cut by some $k$-partition $Y  \in {\cal F}_k$,
$$\frac{1}{2} \tr Y^tLY$$

THE PROBLEM:

$$(k - GP)
\begin{array}{ccc}
 \min & \frac{1}{2} \tr Y^tLY &(=\tr LYY^t)\\
    \mbox{subject to} & Y \in {\cal F}_k
\end{array}
$$

\end{slide}
\begin{slide}{}

$${\cal T}_k  := \lbrace X: X = YY^t   
       \mbox{~for some~}  Y \in {\cal F}_k \rbrace$$

$${\cal E}_m  :=\lbrace X: X=X^t, \diag(X)=u_n, Xu_n = mu_n \rbrace$$

note $X \in {\cal T}_k$ implies that the only
eigenvalues of $X$ are 0 and $m$\\
 Therefore $mI -X \succeq 0$


THE SDP RELAXATION:
$$
\min \lbrace \frac{1}{2} \tr LX : X \in {\cal E}_m,~X \succeq 0,~
mI- X \succeq 0 \rbrace.$$

We can add further constraints, e.g. $X \geq 0$ and other 'polyhedral
constraints'
\end{slide}
\begin{slide}{}
Numerical tests by:\\
 Stefan Karisch and Franz Rendl
\tiny
\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$n$ & ~~$|E|$~~ & ~~$|E_{cut}|$~~ & $(k-GP_{R1})$ & $(k-GP_{R4})$  \\ \hline
36a & 297 & 117 & 111.76 & 116.22 \\
36b & 303 & 118 & 111.63 & 117.18 \\
36c & 316 & 124 & 119.48 & 123.34 \\ \hline

60a & 885 & 367 & 354.46 & 364.84 \\
60b & 863 & 357 & 343.24 & 354.57 \\
60c & 841 & 343 & 329.43 & 341.61 \\ \hline

84a & 1742 & 742 & 716.38 & 734.70 \\
84b & 1793 & 772 & 744.67 & 762.38 \\
84c & 1753 & 753 & 728.01 & 744.25 \\ \hline

108a & 2873 & 1247 & 1205.80 & 1231.12 \\
108b & 2933 & 1283 & 1243.12 & 1264.06 \\
108c & 2871 & 1240 & 1201.78 & 1227.50 \\ \hline

132a & 4294 & 1885 & 1832.76 & 1861.55 \\
132b & 4301 & 1883 & 1842.16 & 1870.04 \\
132c & 4257 & 1854 & 1808.86 & 1837.78 \\ \hline
\end{tabular}
\end{center}
{Partitioning unweighted random graphs into $k=2$ components.}


\begin{center}
\begin{tabular}{ | c | c | c | c c | c |} \hline
$n$ & ~~$|E_{cut}|$~~ &$(k-GP_{R1})$&
\multicolumn{2}{c|}{$(k-GP_{R2})$}&$(k-GP_{R3})$\\ 
\hline
36a & 160 & 149.02 & 154.32 & (102) & 157.36 \\
36b & 163 & 148.84 & 155.84 & (~98) & 159.11 \\
36c & 173 & 159.31 & 166.22 & (103) & 169.40 \\ \hline

60a & 502 & 472.61 & 484.14 & (174) & 489.88 \\
60b & 485 & 457.65 & 468.21 & (185) & 474.03 \\
60c & 470 & 439.24 & 451.49 & (200) & 457.79 \\ \hline

84a & 1011 & 955.18 & ~972.89 & (300) & ~981.17  \\
84b & 1044 & 992.89 & 1007.73 & (290) & 1016.83  \\
84c & 1024 & 970.68 & ~986.38 & (276) & ~993.57  \\ \hline

108a & 1686 & 1607.74 & 1630.80 & (399) & 1642.86 \\
108b & 1738 & 1657.50 & 1675.22 & (376) & 1685.67 \\
108c & 1688 & 1602.38 & 1628.73 & (404) & 1639.40 \\ \hline

132a & 2556 & 2443.69 & 2469.33 & (498) & 2483.07 \\
132b & 2569 & 2456.21 & 2486.40 & (510) & 2498.01 \\
132c & 2526 & 2411.82 & 2442.92 & (527) & 2456.18 \\ \hline
\end{tabular}
\end{center}
{Partitioning unweighted random graphs into $k=3$ components.}

\end{slide}
\begin{slide}{}
\tiny

\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$n$ & ~~$|E_{cut}|$~~&$(k-GP_{R1})$&$(k-GP_{R2})$&$(k-GP_{R3})$\\
\hline
36a & 186 & 167.65 & 179.73 & 180.74 \\
36b & 189 & 167.45 & 182.63 & 183.65 \\
36c & 201 & 179.22 & 193.99 & 195.27 \\ \hline

60a & 577 & 531.69 & 556.13 & 557.92 \\
60b & 558 & 514.86 & 538.32 & 540.09 \\
60c & 540 & 494.14 & 521.45 & 523.07 \\ \hline

84a & 1155 & 1074.57 & 1112.74 & 1114.69 \\
84b & 1195 & 1117.00 & 1152.30 & 1154.58 \\
84c & 1164 & 1092.01 & 1126.55 & 1128.25 \\ \hline

108a & 1918 & 1808.71 & 1860.94 & 1863.37 \\
108b & 1981 & 1864.68 & 1907.70 & 1909.83 \\
108c & 1923 & 1802.68 & 1858.23 & 1860.17 \\ \hline

132a & 2910 & 2749.15 & 2810.40 & 2812.43 \\
132b & 2921 & 2763.24 & 2827.65 & 2829.61 \\
132c & 2874 & 2713.30 & 2781.83 & 2784.19 \\ \hline
\end{tabular}
\end{center}
{Partitioning unweighted random graphs into $k=4$ components.}

\begin{center}
\begin{tabular}{ | c | c | c | c | } \hline
$n$ & ~~$|E_{cut}|$~~&$(k-GP_{R1})$&$(k-GP_{R2})$\\
\hline
36a & 216 & 186.27 & 210.22 \\
36b & 221 & 186.06 & 215.53 \\
36c & 234 & 199.13 & 227.62 \\ \hline

60a & 660 & 590.77 & 638.21 \\
60b & 640 & 572.06 & 618.24 \\
60c & 622 & 549.05 & 600.66 \\ \hline

84a & 1313 & 1193.97 & 1265.56 \\
84b & 1355 & 1241.11 & 1310.61 \\
84c & 1325 & 1213.35 & 1279.39 \\ \hline

108a & 2184 & 2009.68 & 2106.69 \\
108b & 2237 & 2071.87 & 2158.07 \\
108c & 2185 & 2002.97 & 2104.26 \\ \hline

132a & 3290 & 3054.61 & 3171.61 \\
132b & 3292 & 3070.26 & 3187.10 \\
132c & 3251 & 3014.78 & 3141.52 \\ \hline
\end{tabular}
\end{center}
{Partitioning unweighted random graphs into $k=6$ components.}

\end{slide}
\begin{slide}{}
\tiny

\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$n$ & ~~$w(E)$~~ & ~~$w(E_{cut})$~~ & $(k-GP_{R1})$ & $(k-GP_{R4})$  \\
\hline
36d & 3055 & 1426 & 1402.76 & 1425.27 \\
36e & 3168 & 1482 & 1462.12 & 1481.12 \\
36f & 3134 & 1454 & 1435.05 & 1453.24 \\ \hline

60d & 8831 & 4151 & 4118.59 & 4150.11 \\
60e & 8833 & 4154 & 4102.88 & 4149.89 \\
60f & 8777 & 4132 & 4078.85 & 4125.43 \\ \hline

84d & 17241 & 8152 & 8057.77 & 8132.57 \\
84e & 17547 & 8327 & 8233.21 & 8296.74 \\
84f & 17473 & 8264 & 8166.24 & 8237.93 \\ \hline

108d & 29153 & 13895 & 13728.41 & 13825.87 \\
108e & 28877 & 13699 & 13550.40 & 13649.46 \\
108f & 28892 & 13709 & 13588.80 & 13678.63 \\ \hline

132d & 43208 & 20581 & 20380.59 & 20514.25 \\
132e & 43220 & 20618 & 20415.68 & 20536.67 \\
132f & 43391 & 20707 & 20488.58 & 20617.09 \\ \hline
\end{tabular}
\end{center}
{Partitioning weighted random graphs into $k=2$ components.}

\begin{center}
\begin{tabular}{ | c | c | c | c c | c | } \hline
$n$ & ~~$w(E_{cut})$~~ &$(k-GP_{R1})$&
\multicolumn{2}{c|}{$(k-GP_{R2})$}&$(k-GP_{R3})$\\
\hline
36d & 1924 & 1870.35 & 1891.39 & (~96) & 1904.92 \\
36e & 2006 & 1949.49 & 1973.55 & (102) & 1987.21 \\
36f & 1974 & 1913.40 & 1944.87 & (111) & 1956.33 \\ \hline

60d & 5609 & 5491.45 & 5533.67 & (182) & 5558.27 \\
60e & 5592 & 5470.51 & 5520.86 & (195) & 5543.31 \\
60f & 5555 & 5438.46 & 5481.36 & (177) & 5507.30 \\ \hline

84d & 10960 & 10743.70 & 10828.31 & (286) & 10860.45 \\
84e & 11181 & 10977.62 & 11034.14 & (322) & 11068.14 \\
84f & 11105 & 10888.33 & 10966.49 & (315) & 11001.26 \\ \hline

108d & 18624 & 18304.55 & 18398.93 & (433) & 18438.50 \\
108e & 18420 & 18067.20 & 18180.35 & (429) & 18223.56 \\
108f & 18442 & 18118.40 & 18219.78 & (413) & 18258.13 \\ \hline

132d & 27659 & 27174.12 & 27325.69 & (564) & 27374.23 \\
132e & 27657 & 27220.90 & 27332.27 & (506) & 27392.74 \\
132f & 27768 & 27318.11 & 27444.35 & (543) & 27500.82 \\ \hline
\end{tabular}
\end{center}
{Partitioning weighted random graphs into $k=3$ components.}

\end{slide}
\begin{slide}{}
\tiny

\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$n$ & ~~$w(E_{cut})$~~ & $(k-GP_{R1})$ & $(k-GP_{R2})$ & $(k-GP_{R3})$ \\
\hline
36d & 2182 & 2104.14 & 2152.81 & 2159.53 \\
36e & 2268 & 2193.18 & 2243.20 & 2250.25 \\
36f & 2241 & 2152.57 & 2212.78 & 2218.76 \\ \hline

60d & 6341 & 6177.88 & 6288.66 & 6278.02 \\
60e & 6344 & 6154.33 & 6259.22 & 6266.31 \\
60f & 6282 & 6118.27 & 6214.29 & 6222.40 \\ \hline

84d & 12421 & 12086.66 & 12260.11 & 12268.31 \\
84e & 12645 & 12349.82 & 12482.87 & 12492.12 \\
84f & 12553 & 12249.37 & 12411.77 & 12421.67 \\ \hline

108d & 21069 & 20592.62 & 20796.31 & 20804.05 \\
108e & 20836 & 20325.60 & 20562.29 & 20571.90 \\
108f & 20859 & 20383.20 & 20599.70 & 20606.26 \\ \hline

132d & 31277 & 30570.89 & 30888.41 & 30894.53 \\
132e & 31273 & 30623.52 & 30885.18 & 30894.99 \\
132f & 31392 & 30732.87 & 31016.60 & 31025.67 \\ \hline
\end{tabular}
\end{center}
{Partitioning weighted random graphs into $k=4$ components.}

\begin{center}
\begin{tabular}{ | c | c | c | c | } \hline
$n$ & ~~$w(E_{cut})$~~ & $(k-GP_{R1})$ & $(k-GP_{R2})$ \\
\hline
36d & 2454 & 2337.93 & 2434.02 \\
36e & 2552 & 2436.87 & 2533.87 \\
36f & 2525 & 2391.75 & 2502.79 \\ \hline

60d & 7116 & 6864.32 & 7039.53 \\
60e & 7115 & 6838.14 & 7033.36 \\
60f & 7048 & 6798.08 & 6985.54 \\ \hline

84d & 13923 & 13429.63 & 13740.85 \\
84e & 14171 & 13722.03 & 13983.97 \\
84f & 14093 & 13610.41 & 13909.54 \\ \hline

108d & 23575 & 22880.69 & 23258.89 \\
108e & 23303 & 22584.00 & 23007.83 \\
108f & 23345 & 22648.00 & 23043.71 \\ \hline

132d & 34985 & 33967.66 & 34518.13 \\
132e & 34965 & 34026.13 & 34522.69 \\
132f & 35109 & 34147.64 & 34663.42 \\ \hline
\end{tabular}
\end{center}
{Partitioning weighted random graphs into $k=6$ components.}


\end{slide}
\begin{slide}{}
\tiny

\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$k$ & gap$_{R1}$ & gap$_{R2}$ & gap$_{R3}$ & gap$_{R4}$ \\
\hline
2 & 3.34 &  --  &  --  & 0.73 \\
3 & 5.54 & 3.50 & 2.46 &  --  \\
4 & 7.34 & 3.41 & 3.17 &  --  \\
6 & 9.88 & 3.29 &  --  &  --  \\ \hline
\end{tabular}
\end{center}
{Average gaps in percent for unweighted graphs.}

\begin{center}
\begin{tabular}{ | c | c | c | c | c | } \hline
$k$ & gap$_{R1}$ & gap$_{R2}$ & gap$_{R3}$ & gap$_{R4}$ \\
\hline
2 & 1.14 &  --  &  --  & 0.23 \\
3 & 2.07 & 1.31 & 0.95 &  --  \\
4 & 2.66 & 1.22 & 1.12 &  --  \\ 
6 & 3.54 & 1.15 &  --  &  --  \\ \hline
\end{tabular}
\end{center}
{Average gaps in percent for weighted graphs.}


\end{slide}
\begin{slide}{}
\tiny

\begin{center}
\begin{tabular}{ | c | c | c c | c | c | } \hline
$n$ & ~$|E|$~ & 
\multicolumn{2}{c|}{~~$|E_{cut}|$~~} & $(k-GP_{R1})$ & $(k-GP_{R4})$  \\
\hline
124a &  149 &  13 & (~13) &   7.34 &  12.00 \\
124b &  318 &  63 & (~63) &  46.86 &  61.01 \\
124c &  620 & 178 & (179) & 153.01 & 170.91 \\ 
124d & 1271 & 449 & (449) & 418.98 & 440.06 \\ \hline

250a &  331 &  29 & (~29) &  15.44 &  24.87 \\
250b &  612 & 114 & (116) &  81.86 & 100.45 \\
250c & 1283 & 357 & (361) & 303.53 & 327.88 \\
250d & 2421 & 828 & (833) & 747.32 & 779.55 \\ \hline
\end{tabular}
\end{center}
{Partitioning graphs of Johnson et al.\ into $k=2$ components.}

\begin{center}
\begin{tabular}{ | c | c | c | c | c c | c | } \hline
$n$ & $k$ & ~$|E_{cut}|$~&$(k-GP_{R1})$&
\multicolumn{2}{c|}{$(k-GP_{R2})$}&$(k-GP_{R3})$\\
\hline
124a & 4 &   23 &  11.01 &  13.65 & (638) &  17.16 \\
124b & 4 &  107 &  70.29 &  81.36 & (604) &  92.06 \\
124c & 4 &  294 & 229.52 & 251.51 & (704) & 260.65 \\
124d & 4 &  726 & 628.48 & 662.64 & (749) & 669.14 \\ \hline

250a & 5 &   63 &   24.71 & \multicolumn{2}{c|}{  31.08} &   31.08 \\
250b & 5 &  218 &  130.99 & \multicolumn{2}{c|}{ 150.32} &  152.82 \\
250c & 5 &  648 &  485.66 & \multicolumn{2}{c|}{ 527.12} &  530.82 \\
250d & 5 & 1447 & 1195.72 & \multicolumn{2}{c|}{1268.17} & 1269.03 \\ \hline
\end{tabular}
\end{center}
{Partitioning graphs of Johnson et al.\ into $k=4$ or $k=5$ components.}

\begin{center}
\begin{tabular}{ | c | c | c | c | c | c | c | } \hline
$n$ & $k$ & ~$|E_{cut}|$~&$(k-GP_{R1})$&
$(k-GP_{R2})$&$(k-GP_{R3})$ &$(k-GP_{R4})$\\
\hline
120 & 2 &  8 & 1.67 &  --   &   --  & 4.83 \\
120 & 3 & 10 & 2.22 &  4.64 &  7.67 &  --  \\
120 & 4 & 22 & 2.50 &  9.07 & 12.07 &  --  \\
120 & 5 & 20 & 2.67 & 14.26 & 16.50 &  --  \\
120 & 6 & 36 & 2.78 & 22.38 & 25.90 &  --  \\ \hline
\end{tabular}
\end{center}
{Partitioning a geometric graph of size $n=120$ with $|E| =
413$.}

\end{slide}
\begin{slide}{}
\begin{center}
\begin{tabular}{ | r | r | r | r | } \hline
$n$  & $|E|$ & cut & ${\cal E}_m \cap {\cal P}$ \\ \hline
36 & 305   & 119 & 112  \\
60 & 903   & 367 & 355  \\
84 & 1762  & 747 & 717  \\
108 & 2897 & 1252 & 1206 \\
132 & 4325 & 1901 & 1833 \\
\hline
\end{tabular}
\end{center}
Partitioning random graphs into 2 components of equal size.
The first two columns describe the size of the graphs, column 3 contains
the best equipartition found, column 4  contains the lower bound.

\end{slide}
\begin{slide}{}
\begin{center}
\begin{tabular}{ | r | r | r | r | r |} \hline
$n$  &  cut & ${\cal E}_m \cap {\cal P}$ &
${\cal E}_m \cap {\cal P}\cap{\cal N}$ & sign constr. \\ \hline
36 &   160 & 149.1 & 154.3 & 120 \\
60 &   506 & 472.6 & 484.1 & 223 \\
84 &   1014 & 955.1 & 972.8 & 349 \\
108 &  1693 & 1607.7 & 1630.8 & 469 \\
132 &  2563 & 2443.7 & 2469.3 & 554 \\
\hline
\end{tabular}
\end{center}
Partitioning random graphs into $k=3$ components of equal size.
The first  column identifies the graph, column 2 contains
the best 3-partition found, columns 3 and 4 contain lower bounds from
semidefinite relaxations, the last column indicates the number of
nonnegativity constraints $x_{ij} \geq 0$ used, to insure $X \geq 0.$

\end{slide}
\begin{slide}{}
\begin{center}
\begin{tabular}{ | r | r | r | r | r |} \hline
$n$  &  cut & ${\cal E}_m \cap {\cal P}$ &
${\cal E}_m \cap {\cal P}\cap{\cal N}$ & sign constr. \\ \hline
36 &   186 & 167.6 & 179.7 & 211 \\
60 &   585 & 531.7 & 556.1 & 405 \\
84 &   1165 & 1074.6 & 1112.7 & 596 \\
108 &  1935 & 1808.7 & 1860.9 & 859 \\
132 &  2915 & 2749.2 & 2810.4 & 993 \\
\hline
\end{tabular}
\end{center}
Partitioning random graphs into $k=4$ components of equal size.
The first  column identifies the graph, column 2 contains
the best 4-partition found, columns 3 and 4 contain lower bounds from
semidefinite relaxations, the last column indicates the number of
nonnegativity constraints $x_{ij} \geq 0$ used, to insure $X \geq 0.$


\label{endsect:GP}
\label{endsect:QQPSDP}

\end{slide}
\begin{slide}{}


\subsection{Trust Region Subproblems for NLP}
\subsubsection{Background}

Let
\[ q(x):= x^TAx - 2a^Tx,
\]
\begin{eqnarray*}
(TRS)~~~~~ \mu^* := &\min& q(x)\\
&\mbox{s.t.}&
x^Tx = s^2~~(\leq s^2).
\end{eqnarray*}

where

$A=A^T$, not necessarily semidefinite\\
$a \in \Rn$, $s>0.$


\end{slide}
\begin{slide}{}

{\bf Motivation}

Newton's method for minimization - global convergence difficulties

get around these difficulties by minimizing at each iteration
the same quadratic model as in Newton's method,
but restricted to a ball, the \emph{trust
region}. 
~~\\
~~\\

{\bf Main Result:} 
an algorithm that solves TRS using only matrix-vector multiplication; therefore
exploits sparsity; SDP framework


\end{slide}
\begin{slide}{}

{\bf Semidefinite Programming Framework}

\[
(SDP)
\begin{array}{cc}
\max & \trace CZ\\
\mbox{s.t.} & L(Z) = B\\
  & Z \succeq 0
\end{array}
\]

$Z \succeq 0$ denotes positive semidefinite\\
$L$ is a linear operator

\end{slide}
\begin{slide}{}

Back to TRS:
\begin{eqnarray*}
(TRS)~~~~~ \mu^* := &\min& q(x)\\
&\mbox{s.t.}&
x^Tx = s^2~~(\leq s^2).
\end{eqnarray*}

nonlinear least squares, Levenberg 1944, Marquadt 1963

Applications to general minimization, Goldfeld, Quandt and Trotter 1966

Early theoretical results, Forsythe and Golub 1965

Hebden 1973,  Reinsch 1967

efficient numerics, Gay 1981

state of the art algorithm, More-Sorenson 1983

\end{slide}
%\begin{slide}{}
%% start figure 1
%    \begin{figure}[htb]
%      \centering
%      \centerline{\
%\psfig{figure=trpic.ps,height=4in,width=5.9in}}
%      \label{fig1}
%    \ecfigure{Geometric Graph with Bisection}
%%  end figure 1
%\end{slide}
\begin{slide}{}
Currently:

strong Lagrangian Duality, Stern and Wolkowicz 1993

polynomial time, Ye 1992

at most one local-nonglobal, Martinez 1993

two trust regions, CDT 1984

multiple trust regions: Shor, Ramana, Wolkowicz

Lanczos-eigenvalue approach:\\
 Sorensen 1994,\\
(with SDP) Rendl and Wolkowicz 1994,\\
Santos and Sorensen 1995\\
Charles Fortin, (thesis) 2000
\end{slide}
\begin{slide}{}
\[ L(x,\lambda) = q(x) - \lambda(x^Tx-s^2) \]

{\bf Theorem}(Stern and Wolkowicz, 1993)\\
(i) Strong duality holds for TRS, i.e.
\[ \mu^* =  \min_x \max_\lambda L(x,\lambda) = \max_\lambda \min_x
L(x,\lambda).
\]
Moreover, attainment holds for $x$ and uniquely for $\lambda$.\\
(ii) A dual problem for TRS, without a duality gap, is
\[ (D)  \begin{array}{c}
     \max_{A-\lambda I \succeq 0} h(\lambda),
         \end{array}
\]
where 
\[h(\lambda) = \lambda s^2 - a^T(A-\lambda I)^\dagger a, ~(\mbox{concave})
\]
and $\cdot^\dagger$ denotes the Moore-Penrose generalized inverse.
\end{slide}

\begin{slide}{}
{\bf Proof}
\begin{eqnarray*}
\mu^* & = & \min\limits_{||x||=s} q(x) \nonumber\\
      & \geq & \max_\lambda \min_{x} L(x,\lambda) \nonumber\\
      & =& \max\limits_{A-\lambda I \succeq 0} \min_{x} L(x,\lambda),
\end{eqnarray*}
hidden constraints - Hessian $\succeq 0$ and stationarity\\
global minimum of the Lagrangian is
\[ x_\lambda := (A-\lambda I)^\dagger a,
\]
Therefore, the right hand side of the above is equal to
\begin{eqnarray*}
   & =& \max\limits_{A-\lambda I \succeq 0}
         \lambda s^2 -a^T(A-\lambda I)^\dagger a\\
   & =& \max\limits_{A-\lambda I \succeq 0} h(\lambda),
\end{eqnarray*}
This functional is strictly concave and coercive.\\
$h^\prime = 0$ feasibility

\end{slide}
\begin{slide}{}
{\bf Example} (hard case)
Let
\[ A = \left[ \begin{array}{cc}
1 & 0 \\
0 & -1
\end{array}   \right],~a=(1~0)^T,~s=1.
\]
$A-\lambda I \succeq 0,$ implies $\lambda \leq -1$

$\lambda^* = -1$,  $\mu^* = -1.5 = h(\lambda^*)$

$x_\lambda = (A-\lambda I)^{\dagger} a = (.5 ~ 0)^T$

$q(x_\lambda) = -.75 > \mu^*$ 

$||x_\lambda|| < 1$

complementary slackness fails

zero duality gap but no primal attainment
\end{slide}
\begin{slide}{}
{\bf Nonlinear Primal-Dual Pair}

The dual problem is
\[ \begin{array}{c}
     \max_{A-\lambda I \succeq 0} h(\lambda)
         \end{array} ~~~(D) 
\]
where 
\[h(\lambda) = \lambda s^2 - a^T(A-\lambda I)^\dagger a
\]

dual of the dual:
\[
\begin{array}{cc}
\min & h(\lambda)+ \tr Z(A-\lambda I) \\
\mbox{s.t.} & s^2-||(A-\lambda I)^\dagger a ||^2 -\trace Z = 0 \\
      & Z \succeq 0.
\end{array}~~~
(DD)
\]

For feasible pair $Z,\lambda$, the duality gap is:

\[\tr Z(A-\lambda I)~ 
\]
complementary slackness
\end{slide}
\begin{slide}{}
{\bf Gale and More-Sorenson Algorithm\\ explained by SDP Framework}

maintain 2nd order cond. $(A-\lambda I) \succ 0$\\
maximize the dual function $h(\lambda)$\\
 ~~ (Newton's method)

solve $0=h^\prime (\lambda)=s^2-||x_\lambda||^2$\\
better solve $0=\frac 1s - \frac 1{||x_\lambda||}$\\
(almost linear, convex)

Cholesky factoriz: $R^TR=A-\lambda I$ \\
   ~~~~~~~~~~~(safeguard positive def.)\\
Solve~~~~~~~~~~~~~~: $R^TRp=a$\\
Solve~~~~~~~~~~~~~~: $R^Tq=p$\\
iterate~~~~~~~~~~~~: $\lambda \leftarrow \lambda + 
   \left( \frac {||p||}{||q||}\right)^2
   \left( \frac {||p||-s}{s}\right)
     $
\end{slide}
\begin{slide}{}
{\bf Hard Case}

indicated by 
$0 < s^2-||x_\lambda||^2$ ~~inside TR\\

DD feasible 
\[s^2-||(A-\lambda I)^\dagger a ||^2 -\trace Z = 0  \]

so take a primal step to boundary and try to minimize the
duality gap with
\[ Z=zz^T,~\trace Z = ||z||^2\]
\[
\begin{array}{rcl}
   q(x+z)&= &(\lambda s^2 - x_{\lambda}^T (A-\lambda I) x_{\lambda} )
           + z^T (A-\lambda I) z  \\
       &=&  h(\lambda) + z^T (A-\lambda I) z.
  \end{array}
\]
\[\min \trace Z(A-\lambda I) = z^T(A-\lambda I)z\]
\[ A-\lambda I = RRc^T  ~\mbox{Cholesky decomp}   \]
(singular value/vector of $R$)
\end{slide}
\begin{slide}{}
Another approach:\\
{\bf Homogenization of TRS}
\begin{eqnarray*}
\mu^* &=& \min\limits_{||x||=s,~y_0^2=1}  x^TAx - 2y_0a^Tx \\
&=& \max\limits_t \min\limits_{||x||=s}  x^TAx - 2y_0a^Tx +ty_0^2-t \\
&=& \max\limits_t \min\limits_{||x||=s,~y_0^2=1}  x^TAx - 2y_0a^Tx
+ty_0^2-t \\
&=& \max\limits_t \min\limits_{||x||^2+y_0^2=s^2+1}  x^TAx - 2y_0a^Tx
+ty_0^2-t
\end{eqnarray*}

\[ =  \max\limits_t (s^2+1)\lambda_1(D(t)) -t\]

$$D(t) = \left(
\begin{array}{cc}
t & -a^T \\
-a & A
\end{array}
\right).
$$
\end{slide}
\begin{slide}{}
{\bf unconstrained dual problem to TRS}

$$D(t) = \left(
\begin{array}{cc}
t & -a^T \\
-a & A
\end{array}
\right)
$$

$y=\left( \begin{array}{c} y_0\\ x \end{array} \right)$ 
normalized eigenvector for $\lambda_{\min} D(t)$
\[
k(t) =  (s^2+1)\lambda_{\min}(D(t)) -t,
\]

\[  \mbox{***   }~~~ \max_t k(t)
\]

Note
\[  k^\prime(t) = (s^2+1)y_0^2 -1=0  \]
is feasibility again for $x$
\end{slide}
\begin{slide}{}
{\bf SDP Primal-Dual Pair}
\[
\max_t k(t) =  (s^2+1)\lambda_{\min}(D(t)) -t,
\]


add the variable $\lambda$
\[
\begin{array}{cc}
\max & (s^2+1)\lambda - t \\
\mbox{s.t.} & D(t) \succeq \lambda I
\end{array}
(DSDP)
\]

Lagrangian dual of this dual is:
\[
\begin{array}{cc}
\min & \tr D(0)X \\
\mbox{s.t.} & \tr X = s^2+1 \\
      &  X_{11} = 1\\
      & X \succeq 0
\end{array}(PSDP)
\]
\end{slide}
\begin{slide}{}
primal-dual interior point method:

approx. solve perturbed optimality conditions
using Newton's method:
\[
\begin{array}{c}
  \tr X = s^2+1 \\
        X_{11} = 1\\
  D(t) - \lambda I - Z = 0\\
  \mu Z^{-1} - X = 0  \\
       X \succ 0, Z \succ 0
\end{array}
\]


\end{slide}
\begin{slide}{}
Dual Simplex Method

basic feasible dual solution at given $t$:
\[ D(t) - \lambda_{\min}(D(t)) I  \succeq 0 ~ \mbox{and singular}\]

eigenvector $ y=\left( \begin{array}{c} y_0\\x \end{array} \right) $

normalize $y_0=1$

test primal feasibility of $X=yy^T$ to get better value for $t$ (interpolation problem)

Alternatively, normalize $||y||=1$ with $y_0 \geq 0$,\\
and exploit almost linear structure of
\[
\frac 1{y_0(t)} = \sqrt{s^2+1}
\]
\end{slide}
\begin{slide}{}
\[ Z = D(t) -\lambda I
\]
complementary slackness
\[ \trace ZX = 0  \]
\[  0=ZX=
\left[  \begin{array}{cc}
       t-\lambda  & -a^T \\
      -a          & A-\lambda I  \end{array} \right]
\left[  \begin{array}{cc}
       X_{11} & y^T\\
      y         & \bar{X}  \end{array} \right].
\]

\[
   t=a^Ty+\lambda;
\]
\[
   \bar{X}a=(t-\lambda)y;
\]
\[
   (A-\lambda I)y=a;
\]
\[
   (A-\lambda I)\bar{X} = ay^T;
\]
\[
   A-\lambda I \succeq 0.
\]

\end{slide}
\begin{slide}{}
\[ \lambda_i (A) = \alpha_i  \]

$J := \lbrace i: a_i \not= 0 \rbrace$

$  \det (D(t) - \lambda I)=$ 
\begin{eqnarray*}
&=&(t- \lambda ) \prod_{k=1}^n (\alpha_k -
\lambda ) - \sum_{k=1}^n a_k^2 \prod_{j \not= k}^n (\alpha_j - \lambda )\\
&=&\lbrack t- \lambda  -
\sum_{j \in J} \frac{a_j^2}{\alpha_j - \lambda} \rbrack
\prod_{k=1}^n (\alpha_k - \lambda )\\
&=&d(\lambda)
\prod_{k=1}^n (\alpha_k - \lambda )
\end{eqnarray*}

\end{slide}
\begin{slide}{}
Easy case:
\begin{enumerate}
\item[]
$$\lambda_1 (D(t)) \mbox{~is simple and~} \lambda_1 (D(t)) <
\alpha_1$$
\item[]
if $\lambda < \alpha_1$ given. Then
\[  D(d(\lambda)) - \lambda I \succeq 0 ~ \mbox{and singular}.
\]
\item[]
Let $y(t)$ be a
normalized eigenvector corresponding to $\lambda_1(D(t))$ and denote its
first component by $y_0(t)$. Then $y_0(t) \not= 0$. (wlog $>0$)
\item[]
$y_0(t) : \RRc \mapsto (0,1)$ is strictly
monotonically decreasing.
\end{enumerate}
\end{slide}
\begin{slide}{}
{\bf Theorem}\\
$y(t)=\left(\begin{array}{c} y_0(t)\\ z(t) \end{array}
        \right)$ normal eigenvector for 
\[\lambda_1(D(t))\]
Then\\
$y_0(t) \not= 0$ and $v := \frac{1}{y_0(t)}z(t)$
is unique opt of
\[
\min \lbrace v^TAv -2a^Tv: v^Tv = \frac{1 - y_0(t)^2}{y_0(t)^2} \rbrace.
\]

Conversely, if $v \in \Rn ~and~ \lambda \in \RR$ satisfy\\
\[(A - \lambda I)v = a; ~~A - \lambda I \succeq 0;~
v^Tv = \frac{1-y_0^2}{y_0^2}\]
thereby defining $ y_{0} > 0,$\\ 
then\\
 $y := y_0\left(\begin{array}{c}1\\ v \end{array}
        \right)$\\
 is eigenvector for $D(t)$ for $t := a^Tv + \lambda$\\
 and \[\lambda_1(D(t)) = \lambda\]
\end{slide}
\begin{slide}{}
\begin{center}  {\bf Dual-Simplex Method}   \end{center}
\begin{tabbing}
123\=456\=789\=\kill
{\bf Initialization:}\\ 
Find bounds for intervals of uncertainty.\\
$ t_0^l \leq t^* \leq t^u_0$\\
$\mu^l \leq \mu^* \leq \mu^u$\\
~~\\
Begin the main iterations:\\
{\bf while} (feasibility or duality gap tols too large)
~~\\
\> Part 1: Find a new estimate $t_+$.\\
\> $t_+ = \frac{t_c^l+t_c^u}{2}$  (default value) \\
\> \> Find intersection point of 2 tangent lines\\
\> \> ~~~~~to the graph of $k(t)$\\
\> \> update the upper bound $\mu^u$\\
\> {\bf if} last iterate, on infeasible side, good side\\ 
\> \> use inverse interpolation\\
\> {\bf elseif} on feasible side, bad side\\
\> \> use inverse interpolation\\
\> {\bf endif}\\
\> Part 2: Update info for new estimate $t_+$.\\
\> {\bf if} $\lambda > 0$  (wrong sign for Lagrange multiplier)\\
\> \> {\bf if} $t_+ <= t^*$  then {\bf STOP}\\
\> {\bf else}   (correct sign for Lagrange multiplier)\\
\> \> Update lower bound using dual function\\
\> {\bf endif}\\
\> {\bf if} $t_+<t_*$  ( bad side)\\
\> \> Perform a PSDP (negative curvature) \\
\> \> ~~~~step, update bounds \\
\> {\bf else}\\
\> \> steepest descent step - update bounds\\
\> {\bf endif}\\
\> Update the tolerances\\
{\bf endwhile}
\end{tabbing}
\end{slide}
\begin{slide}{}
Tests on SUN SPARC station 1 using MATLAB
\begin{enumerate}
\item
dimensions 1540 to 1565  \\
     30 problems for each dimension  \\
     density of nonzeros .01  \\
     average iterations   4.4453  \\
     average cpu time 54.1876 sec  \\
     average work 1.73 times one Lanczos step\\

\item
dimensions 1540 to 1558  \\
     30 problems for each dimension  \\
     density of nonzeros .013 \\
     average iterations   4.1148  \\
     average cpu time 56.1322 sec  \\
     average work 1.75 times one Lanczos step\\
     comment: a multiple of the identity was added to get a positive
            definite  Hessian

\newpage
\item
dimensions 1500 to 1565  \\
     2 problems for each dimension  \\
     density of nonzeros .01  \\
     average iterations   4.2951  \\
     average cpu time 57.0429 sec  \\
     average work 4.8589 times one Lanczos step\\
     comment: hard case; multiplicity of smallest eigenvalue varied from 1 to 6

\item
dimensions 1800 to 1865  \\
     12 problems for each dimension  \\
     density of nonzeros .005  \\
     average iterations   4.2453  \\
     average cpu time 28.1692 sec  \\
     average work 4.2589 times one Lanczos step\\
     comment: hard case; multiplicity of smallest eigenvalue varied from 1 to 6
\end{enumerate}

\end{slide}
\begin{slide}{}

\subsection{Sequential Quadratic Programming for NLP}

$\bullet$SQP uses a quadratic program to find the search direction, i.e.
quadratic approximation of objective and linear approximations of
constraints.

$\bullet$Extend the classical SQP method to include a quadratic model of the
constraints. Solve the Lagrangian relaxation of this QQP subproblem
using SDP.

Until now we have used classical tools of nonlinear programming to
develop and analyse a modern problem in linear optimization.  In a
reversal of roles, we now attempt to use semidefinite programming
as the subproblem solver of a nonlinear optimization toolbox. This
should be viewed as an application of semidefinite programming. 
The results presented here have been published before. 

A proven approach for the unconstrained minimization of a function,
${f(x)},{x\in\Rn},$ is to build and solve a quadratic model at a local
estimate $\xk,$ that is, apply the Newton's method.  In this chapter
we propose a direct extension of this modeling approach to constrained
minimization: A local quadratic model of both the objective function
and the constraints is built; since this model is too hard to solve,
it is relaxed using the Lagrangean dual, which is then solved by
semidefinite programming techniques. The key idea in this approach is
to use the latest technique of cone linear programming to obtain a
better model than is usual in SQP methods and the key ingredient is
the equivalence between the Lagrangean and semidefinite relaxations.

As illustration of how semidefinite programs is used to good
effect, recall the well-known Rayleigh-Ritz quotient to obtain the
smallest eigenvalue of a symmetric matrix $A$. An equivalent
formulation yields (for example, see \cite{HoJo:85})
\begin{equation}\label{rayleigh-ritz}
  \lambda_1(A) = \minpgms{x^TAx}{x^Tx=1}.
\end{equation}
One approach to prove this result involves Lagrange multipliers:
the optimal $x$ must be a stationary point of the Lagrangean
$L(x,\lambda)=x^TAx-\lambda(x^Tx-1).$ This shows that the optimal $x$
is an eigenvector; and substitution into the objective function shows
that the corresponding eigenvalue is the smallest. But now, consider
instead, $x^TAx = \trace{(x^TAx)} = \trace{(Axx^T)}$ and let
$X:=xx^T$.  We write the program (\ref{rayleigh-ritz}) as
\begin{displaymath}
  \minpgms{\inner{A}{X}}{\inner{I}{X}=1,X\succeq 0,X=xx^T},
\end{displaymath}
where $\inner{A}{B}=\tr(AB)$, the trace inner product; $A \succeq 0$
(resp. $A \succ 0$) denotes positive semidefiniteness (resp. positive
definiteness); and $A \succeq B$ denotes $A-B \succeq 0$, i.e. the
symmetric matrix space $\Sn$ is equipped with the L{\"{o}}wner partial
order.

Note that the rank one constraint ($X=xx^T$) is redundant because we
have only one constraint \cite{GPat:95}.  We therefore drop it and
construct the dual to obtain
\begin{displaymath}
  \maxpgms{\lambda}{\lambda I \preceq A,\lambda\in\RE},
\end{displaymath}
which obviously has $\lambda_1(A)$ as optimal value.  Since the dual
has a strictly interior point, the primal attains the same value and
we get the Rayleigh-Ritz result. The reader should concur that this
has to be one of the simplest proof of (\ref{rayleigh-ritz}).  In this
manner we use semidefinite programming to solve Lagrangean
relaxations. 

In this chapter, we wish to illustrate some of the strengths, both
theoretical and practical, of considering semidefinite relaxations of
quadratic programs as the tool of choice for solving Lagrangean
relaxations that arise from quadratic models of general nonlinear
programs. 

\subsubsection{The Simplest Case}
\label{sect:simple}
Moving up in complexity we consider the unconstrained problem
\begin{displaymath}\label{uncpgm}
  \minpgms{f(x)}{x\in\Rn}.
\end{displaymath}
When possible, the method of choice for this problem is Newton's
method, which solves a quadratic model of the objective function.  To
ensure a solution (or convexity) of the model, Newton's method is
often implemented within a Trust-Region, or Restricted-Step approach.
This very efficient variation proceeds from an initial estimate of
the solution; develops a second-order model of the objective
function deemed valid in a region around the estimate; and finally solves the
model (the trust-region subproblem)
\begin{equation}\label{trspgm}
  \minpgms{q_0(d):=d^TQd + 2b^Td }{ q_1(d):= d^Td \le \delta^2, d\in \Rn}.
\end{equation}
The model is constructed from $Q=\nabla^2f(\xk)$ (or an approximation
of the Hessian), $b=\nabla f(\xk)$ and the parameter $\delta$
represents the radius where the model is deemed valid. The trust-region may
be scaled or even arise from a non-convex quadratic.  A solution $d$ is
then used as the step to the next estimate $\xkp=\xk+ d$.

One of the interesting properties of (\ref{trspgm}), first shown in Stern and
Wolkowicz \cite{StWo:93} using semidefinite programming, is that even
though generally non-convex, the problem exhibits no duality gaps. The
Lagrangean dual of (\ref{trspgm}) is written as
\begin{equation}\label{pgm:nlind}
  \maxpgms{-b^T(Q+\lambda I)^\dagger b-\lambda\delta^2}
        {Q+\lambda I \succ 0,\lambda\ge 0},
\end{equation}
a nonlinear, concave semidefinite program, where $(\cdot)^\dagger$ is
the Moore-Penrose generalized inverse.  In addition, the Lagrangean
dual has been shown \cite{ReWo:94} equivalent to the following
linear semidefinite program,
\begin{equation}\label{pgm:lind}
  \maxpgms{(\delta^2+1)\lambda - t} { 
    \left[ 
      \begin{array}{cc} t & b^T \\ b & Q
      \end{array}
    \right] 
    \succeq \lambda I, t\in\RE, \lambda \geq 0 }.
\end{equation}
We take the dual of the above linear semidefinite program (\ref{pgm:lind})
to get a semidefinite program equivalent to (\ref{trspgm}).
\begin{equation}\label{pgm:linp}
  \minpgms{\inner{P_0}{Y}}{ \inner{E_{0}}{Y}=1,
    \inner{P_I}{Y} \le \delta^2, Y \succeq 0}.
\end{equation}
The variable in this program, $Y$, belongs to the cone of symmetric
positive semidefinite matrices of dimension $(n+1) \times (n+1).$ Also,
\begin{displaymath}
  P_0=\left[\begin{array}{cc}0&b^T\\b&Q\end{array}\right],
  P_I=\left[\begin{array}{cc}0&0\\0&I\end{array}\right],
  E_{0}=\left[\begin{array}{cc}1&0\\0&{\bf 0}\end{array}\right].
\end{displaymath}
The reader will note that (\ref{pgm:linp}) may be obtained as we did
for the for the Rayleigh-Ritz program, by homogenization of
(\ref{trspgm}), transformation to matrix space and then by dropping
the rank one constraint. (We abuse the term homogenization to mean a
quadratic function without a linear term.) We will do this in detail
for a more general program later on.

This pair of linear primal-dual semidefinite programs (\ref{pgm:linp},
\ref{pgm:lind}) have strict interior points. Therefore the optimal
values are equal; moreover, they are both attained.  Finally, part of
the first column of the primal semidefinite solution, the matrix $Y$,
is feasible for (\ref{trspgm}).  And, possibly with an additional
displacement, chosen in the nullspace of the Lagrangean, this first
column yields the same objective value for (\ref{trspgm}) as its dual
optimal.  By this procedure, usually known as {\it lifting}, of
(\ref{trspgm}) to the cone of semidefinite matrices, and projecting
back (by the first column), we see that there are no duality gaps for
(\ref{trspgm}).  This was first shown in \cite{StWo:93}.
\begin{thm}
  The optimal solution to (\ref{trspgm}) and to its Lagrangean dual problem
  (\ref{pgm:nlind}) are attained and the corresponding objective values are
  equal.
\end{thm}
The interesting aspect of this theorem is that the Lagrangean dual is
shown equivalent to a semidefinite program and its optimal value is
deduced from this latter program.  Therefore, interior-point
algorithms, as developed in the previous chapters, may be used to
solve (\ref{trspgm}), even if the objective function and the feasible
set are non-convex. The result has been extended to upper and lower
bounded trust-region subproblems but, interestingly, not to a finite
number of constraints.  With as few as two constraints, a duality gap
may appear \cite{MR91e:90078,Yuan:91}.


\subsubsection{Multiple Trust-Regions}
\label{sect:multiple}
Consider now a quadratic objective function constrained by multiple
quadratics,
\begin{equation}\label{pgm:qqp}
\minpgms{x^T Q_0 x + 2b_0^T x - a_0}{
        x^T Q_k x + 2b_k^T x - a_k \le 0,1\le k \le m,x\in\Rn}.
\end{equation}
As soon as two or more trust-regions are considered, the necessary and
sufficient conditions that hold for one trust region may no longer be
necessary for (\ref{pgm:qqp}).  This is reflected in the duality gap
exhibited by some instances of multiple trust-region programs.

To directly derive the relaxations, we introduce the vector $y=(x_0\ 
x)^T$. We then require $x_0^2=1$ or, in terms of the new variable,
$y^TE_{0}y=1$, to get an homogeneous program equivalent to
(\ref{pgm:qqp}),
\begin{equation}\label{pgm:hqqp}
  \minpgms{y^TP_0y}
  {y^TE_{0}y=1,y^TP_ky\le 0,1\le k\le m,y\in\RE^{n+1}},
\end{equation}
where 
\begin{displaymath}
  E_{0} = 
  \left[ 
    \begin{array}{cc} 
      1 & 0 \\ 0 & {\bf 0}
    \end{array} 
  \right]\mbox{ and }
  P_k = \left[ \begin{array}{cc} -a_k & b_k^T \\ b_k & Q_k
  \end{array} \right], 0\le k \le m.
\end{displaymath}
The homogenization simplifies the notation and opens the way to the
semidefinite relaxation: We rewrite (\ref{pgm:hqqp})\ using matrix
variables.
\begin{displaymath}\label{pgm:hmqqp}
  \minpgms{\inner{Y}{P_0}}
  {\inner{Y}{E_{0}}=1,\inner{Y}{P_k}\le 0,1\le k\le
  m,Y=yy^T}.
\end{displaymath}
The rank-one constraint is relaxed to a semidefinite constraint; a
procedure we justify by showing its equivalence with the
Lagrangean relaxation.  After some rearrangement of terms, the
Lagrangean dual of (\ref{pgm:hqqp})\ reads
\begin{displaymath}
  \max \left\{
   \min \biggl\{y^T(P_0+\sum_{k=1}^m\lambda_kP_k
       +\lambda_0 E_{0})y-\lambda_0  \mid
  y\in \RE^{n+1}  \biggr\}  \bigm|
  \lambda \in \RE \times \RE^m_{+} \right\}.
\end{displaymath}
For the inner minimization to be bounded we must now have
\begin{displaymath}
  P_0+\sum_{k=1}^m\lambda_kP_k+\lambda_0 E_{0} \succeq 0,
  \quad
  \mbox{which implies}
  \quad
  Q+\sum_{k=1}^m\lambda_kQ_k \succeq 0.
\end{displaymath}

This, by the way, is where the duality gap arises. The standard
necessary optimality conditions for (\ref{pgm:qqp})\ do not require
the Hessian of the Lagrangean to be semidefinite.  But the Lagrangean
dual program we are deriving here requires the same Hessian to be
semidefinite.  We therefore cannot expect the primal variables
corresponding to an optimal dual solution to be, in general, optimal
for (\ref{pgm:qqp}).

To complete the derivation, we note that the minimum over $y$ will
be attained at $y=0$ from which we get the dual program
\begin{equation}\label{pgm:dqqp}
  \maxpgms{-\lambda_0}{P_0+\lambda_0
    E_{0}+\sum_{k=1}^m\lambda_kP_k\succeq 0,\lambda \in \RE \times \RE^m_{+}}.
\end{equation}

We have now justified the claim of equivalence of the Lagrangean and
semidefinite relaxations since dropping the rank-one condition on the
homogenized primal (\ref{pgm:hmqqp}) or taking the semidefinite dual of
(\ref{pgm:dqqp})\ will result in the following, which we will therefore
simply refer to as the relaxation of (\ref{pgm:qqp}),
\begin{equation}\label{pgm:sdpqqp}
 \minpgms{\inner{P_0}{Y}}{
        \inner{E_{0}}{Y}=1, 
        \inner{P_k}{Y} \le 0, 1 \le k \le m, 
        Y \succeq 0}.
\end{equation}
This resulting semidefinite relaxation of (\ref{pgm:hqqp})\ is
equivalent to the one considered in the literature
(\cite{Sho:87,BoGhFeBa:93,PoReWo:94}).


The optimal value of the relaxation provides a lower bound for the
(\ref{pgm:qqp}). We now need an approximation for the optimum $x$.
Feasibility properties of the first column of the semidefinite
relaxation were first shown by Fujie and Kojima \cite{FuKo:95} for an
equivalent problem with linear objective function.  For an alternate
view of this result, see old geometry KW paper??? from which we extract the
next results.  Consider the feasible set of the nonlinear program (\ref{pgm:qqp}),
\begin{displaymath}
  \hat{F}\df\{x\in\Rn\mid x^T Q_k x + 2b_k^T x - a_k \le 0,1\le k \le m\};
\end{displaymath} 
the feasible set of the semidefinite program (\ref{pgm:sdpqqp}),
\begin{displaymath}
  \tilde{F}\df \{Y \succeq 0 \mid  \inner{E_0}{Y}  = 1,
                              \inner{P_k}{Y}  \le 0, 1\le k \le m\};
\end{displaymath}
and the projector map,
\begin{displaymath}\label{projection}
  P_R\colon\Sn \rightarrow \Rn,\quad P_R(Y)=
  P_R\left(\left[\begin{array}{cc}a&x^T\\x&X\end{array}\right]\right) = x.
\end{displaymath}
\begin{thm} \label{lem:qqp_feasible}
  Suppose that $Y$ is a feasible solution of (\ref{pgm:sdpqqp}). The projected
  vector, $x=P_R(Y)$, is then feasible for all convex constraints of (\ref{pgm:qqp}).
\end{thm}\bpr 
Since we are concerned only with convex constraints, we may consider
only those where $Q_k\succeq 0$ and compute
\begin{eqnarray*}
  x^TQ_kx+2b^T_kx -a_k- \inner{P_k}{Y} &=& x^TQ_kx-\inner{Q_k}{X}\\
  &=&-\inner{Q}{X-xx^T}.
\end{eqnarray*}
Since $Y \succeq 0$ implies $X-xx^T \succeq 0$, we obtain
\begin{eqnarray*}
  x^TQ_kx+2b^T_kx -a_k &=& \inner{P_k}{Y}-\inner{Q_k}{X-xx^T}\\
  &\le& \inner{P_k}{Y}\\
  &\le& 0.
\end{eqnarray*}
And therefore $x$ is feasible for all convex constraints of
(\ref{pgm:qqp}). \epr

This feasibility of the first column is interesting to consider in
more detail.  First, in the case of a problem where the quadratic
constraints are convex (but maybe the objective is not) there is an
obvious way to improve this first column solution when it is not
optimal.

A optimal pair $Y$,$\lambda$ to the semidefinite relaxation, if $Y$ is
not rank one, will in general map to a vector $x$ for which
complementarity fails but improving the objective value while
remaining feasible is then easy.
\begin{lemma} \label{lem:cvxstep}
  Consider a (\ref{pgm:qqp})\ with convex constraints. If the semidefinite
  primal optimal solution $Y$ is not rank one, let $\tilde x =
  P_R(Y)$, (part of the first column of $Y$).  Then there is a $\bar
  x$ chosen in ${\cal N}(Q_0+\sum \lambda_k Q_k)$, the nullspace of
  the Lagrangean, such that $x=\tilde x + \bar x$, is feasible and
  will improve the primal objective value of (\ref{pgm:qqp}).
\end{lemma}
The idea is to choose a displacement along the nullspace of the
Lagrangean until one or more slack constraints is satisfied with
equality. The value of the objective function is lowered since
\begin{displaymath}
  0 = {\bar x}^T (Q_0+\sum \lambda_k Q_k) \bar x \ge {\bar x}^T Q_0
  \bar x
\end{displaymath}
and therefore $(\tilde x + \bar x)^T Q_0 (\tilde x + \bar x) \le
{\tilde x}^T Q_0 \tilde x.$
  
Consider now a more general case where the constraints may not be
convex. Note that Theorem \ref{lem:qqp_feasible} implies that the
projected first column $x$ is feasible for any nonnegative combination
of constraints,
\begin{equation}
  \label{eq:validconv}
  \sum_{k=1}^m\lambda_k(x^T Q_k x + 2b_k^T x - a_k) \leq 0,\lambda\geq 0,
\end{equation}
which results in a convex function.  Thus we obtain feasible
points for convex combinations of constraints of (\ref{pgm:qqp})\ as in
(\ref{eq:validconv}) from feasible points $Y$ of the relaxation
(\ref{pgm:sdpqqp}), even when these are not rank one.  Therefore the
relaxation provides a convex approximation to the feasible set
$\hat{F}$.  However, it actually provides a better approximation than
this would initially lead us to believe.  

Let us define a \textit{valid inequality} for (\ref{pgm:qqp})\ as 
\begin{equation}
  \label{eq:validnonc}
  \sum_{k=1}^m\lambda_k(x^T Q_k x + 2b_k^T x - a_k) \le 0,
  \quad\mbox{where}\quad   Q_0+\sum_{k=1}^m\lambda_kQ_k \succeq 0,
           \lambda \geq 0.
\end{equation}
These inequalities, an infinite number of them, are not, in general,
convex. (Simply consider (\ref{trspgm}) where the objective is strictly
convex while the constraint is not.)  However, they provide geometric
insight into the \sdp\ relaxation.  The set of vectors satisfying all
valid inequalities,
\begin{displaymath}
  \Bigl\{x\Bigm| \sum_{k=1}^m\lambda_k(x^T Q_k x + 2b_k^T x - a_k) \le 0,
  \quad Q_0+\sum_{k=1}^m\lambda_kQ_k \succeq 0, \lambda \geq 0 \Bigr\}
\end{displaymath}
establishes a relation between the set of projected columns of \sdp\ 
solutions and some intersection of the original constraints.

We now use the above geometric descriptions to provide an approximate
solution to (\ref{pgm:qqp})\ from the optimum of \sdp.  We use the first column
of the optimum $Y$ but then we use the properties of the valid
inequalities (\ref{eq:validnonc}) to improve this column by moving
onto a boundary of a valid inequality.

In the general case of a non-convex feasible region, we obtain a
step, which, unlike Lemma \ref{lem:cvxstep}, {\em attains}
complementary slackness, though not necessarily feasibility. Again,
the value of the objective function is improved.  This additional step
is a generalization of an idea introduced by Mor{\'e} and Sorensen
to solve (\ref{trspgm}) and there is an explicit expression for
the step as there is for (\ref{trspgm}), given here in Lemma (\ref{lem:alpha}).
We give the technical construction of the step in the following lemma
and its value in Corollary \ref{cor:alpha}.

%%%%%% Start of extract from geometry paper
\begin{lemma}
\label{lem:alpha}
Suppose that 
$\lambda$ and 
\begin{displaymath} Y = 
\left[  \begin{array}{cc}
    1 & x^T\\
    x   & X  \end{array} \right]
\end{displaymath}
are feasible for the primal-dual pair DSDP and PSDP, respectively.
Let
\begin{displaymath} 
  y := \left[  \begin{array}{c}
      1 \\
      x   \end{array} \right],\qquad
  Z  := 
  P_0  + \lambda_0 E_{0} + \sum_{k \in {\cal I} } \lambda_k P_k, 
  \quad {\cal I}=\left\{1, \ldots m\right\}
\end{displaymath}
and suppose that they satisfy $ZY=0$.

Let the matrix $Y$ be factored as
\begin{displaymath}  
  Y = T T^T, 
\end{displaymath}
where $T$ is $(n+1) \times r$ and full column rank $r \geq 2.$ Let the
matrix $S$ be $r \times (r-1)$ and full column rank with ${\cal R}(S)=
{\cal N}(T_{1,:}),$ i.e. with range space given by the orthogonal
complement to the first row of $T$.  Define
\begin{displaymath} 
  R := TS,~\bar{P} := \sum_{k \in {\cal I}} \lambda_k P_k,~
  c := y^T\bar{P}y,
\end{displaymath}
\begin{displaymath}
  K := R^T\left(\bar{P}yy^T\bar{P} - c \bar{P}\right)R.
\end{displaymath}
Choose $v$ such that
\begin{equation}
  \label{eq:vtoz}
 Rv \neq 0 ~\mbox{and}~ v^TKv \geq 0,
\end{equation}
 and define
\begin{displaymath}
a := v^TR^T\bar{P}Rv,~b := 2v^TR^T\bar{P}y.
\end{displaymath}
Then, for $z$ defined as follows, we have
\begin{equation}
  \label{eq:tsv}
TSv = 
  \left[  \begin{array}{c}
       0 \\
      z   \end{array} \right]   \neq 0,
\end{equation}
and
\begin{equation}
  \label{eq:discr}
b^2-4ac \geq 0.
\end{equation}
Moreover, if we define
\begin{displaymath} 
  \alpha:=
  \left\{  \begin{array}{cc}
      \left( -b \pm \sqrt{b^2 -4ac}\right)/(2a) & \mbox{if}~a \neq 0 \\
      \frac{-c}b   & \mbox{if} ~ a=0,
    \end{array}
  \right.
\end{displaymath}
and
\begin{displaymath}
  w:= y+\alpha
  \left[  \begin{array}{c}
      0 \\
      z   \end{array} \right],
\end{displaymath}
then 
\begin{equation}
  \label{eq:feaswpw}
  w^T\bar{P}w = 0,~\mbox{and}~ Zw=0.
\end{equation}
\end{lemma}\bpr 
That (\ref{eq:tsv}) holds and $Zw=0$ follows directly from
construction of $R$ and the assumption of complementary slackness,
$ZY=0.$ Note that $ZY=ZTT^T=0$ implies $ZT=0.$ We still need to show
the equality of the quadratic form in (\ref{eq:feaswpw}). Now
\begin{displaymath}
  w^T \bar{P} w = y^T\bar{P}y + \alpha 2v^TR^T \bar{P} y +
  \alpha^2v^TR^T \bar{P} R v.
\end{displaymath}
(We assume that a $w$ exists to make this quadratic 0. This may be seen
from using the ordinary TRS with the given $\lambda$ defining the single
constraint.)
The discriminant for this quadratic in $\alpha$ is defined in
(\ref{eq:discr}), where
\begin{displaymath} 
  b^2-4ac = 4v^T K v.
\end{displaymath}
Therefore, the discriminant is nonnegative, and the quadratic has a
real solution $\alpha$ as given by the standard formula. \epr

We now make explicit the value of the above lemma in finding an
approximate solution to (\ref{pgm:qqp}).
\begin{cor}
\label{cor:alpha}
Suppose that $Y$,$Z$,$\lambda$,$w$ are defined as in Lemma
\ref{lem:alpha} above. Then the Lagrangean dual bound is attained by
$w$ as well as complementary slackness.
\end{cor}\bpr 
That complementarity is attained is seen directly from the second
equation of (\ref{eq:feaswpw}).  And from both equations we obtain
\begin{displaymath}
  0 = w^TZw-w^T\bar{P}w = w^T(P_0 - \lambda_0 E_0)w.
\end{displaymath}
Therefore, $w^TP_0w = q_0(x+\alpha z) = -\lambda_0$, the dual
Lagrangean bound. \epr
%%%%%% End of extract

\subsubsection{Approximations of Nonlinear Programs}
\label{sect:approx}
We assume the reader is familiar with {\it Sequential Quadratic
  Programming}, denoted \sqp. We recall only the main features and
refer the reader to ???? for details.  The
usual justifications for the application of \sqp\ to the nonlinear
program
\begin{displaymath}
  \minpgm{\nep}{f_0(x)}{f_i(x)=0, 1\le i\le m, x\in\Rn},
\end{displaymath}
stem from applying Newton's method
to obtain stationarity of its Lagrangean $\Lag(x,\lambda) \df f_0(x) +
\sum\lambda_i f_i(x)$, 
\begin{eqnarray*}
  \nabla f_0(x^*) + \sum\nabla f_i(x^*)\lambda_i^* = 0,\\ f(x^*) = 0.
\end{eqnarray*}
We will sometimes use the notation $f(x) := [f_1(x) \ldots f_m(x)]^T$
and $f'(x)$ for the first derivative.  An iterative attempt at the
non-linear system above by Newton's method with some simplification
involving $d = \xkp - \xk$ and $\delta_\lambda = \lkp-\lk$, will
produce the First-Order Newton Step,
\begin{displaymath}\label{lab:foc}
  \left[ \begin{array}{cc}
      \nabla^2\Lag(\xk,\lk) & f'(\xk)
      \\ f'(\xk)^T & 0
        \end{array} 
      \right] \left[ \begin{array}{c} d \\ \lkp \end{array}
      \right] = \left[ \begin{array}{c} - \nabla f_0(\xk) \\ - f(\xk)
        \end{array}
      \right].
\end{displaymath}
This system produces a direction $d$ and a new vector of Lagrangean
multiplier estimates $\lkp$.  The key justification for \sqp\ is that
the system of equations (\ref{lab:foc}) may be derived as the first-order
necessary conditions of the quadratic program
\begin{equation}\label{pgm:qp}
\begin{array}{lrlc}
  & \min&f_0(\xk)+\nabla f_0(\xk)^T d+\half d^T
  \nabla^2\Lag(\xk,\lk)d\\ 
  &\mbox{s.t.}& f_i(\xk) + \nabla
  f_i(\xk)^T d =0, \qquad 1\le i \le m.
\end{array}
\end{equation}
Stationarity of the Lagrangean of (\ref{pgm:qp}) yields the first line of (\ref{lab:foc}),
and feasibility yields the second line.  This is why \sqp\ is viewed
as an extension of Newton's method to constrained optimization.  

It is now standard procedure to extend the above derivation to the
inequality constrained program.
\begin{equation}\label{pgm:nlp}
  \minpgms{f_0(x)}{f_i(x)\le 0, 1\le i \le m, x\in\Rn},
\end{equation}
and obtain the subproblem,
\begin{equation}\label{lab:nlpqp}
\begin{array}{lrlc}
  & \min&f_0(\xk)+\nabla f_0(\xk)^T d+\half d^T
  \nabla^2\Lag(\xk,\lk)d\\ 
  &\mbox{s.t.}& f_i(\xk) + \nabla
  f_i(\xk)^T d \le 0, \qquad 1\le i \le m,
\end{array}
\end{equation}

In summary, the above derivation, from the Taylor first-order
expansion of $\Lag(x,\lambda)$, obtained the standard \sqp\ 
subproblem, which approximates the objective function to second order
yet approximates the constraints only to first order.  Consider now a
second-order Taylor expansion of $\Lag(x,\lambda)$,
\begin{displaymath}
  \left[
    \begin{array}{c}
      \sum\lambda_i\nabla f_i(\xk) 
      + \nabla^2\Lag(\xk,\lk) d
      +H_3(\delta_x, \delta_\lambda) \\
      f'(\xk) d + \half d^T f''(\xk) d
    \end{array}
  \right] =
  \left[
    \begin{array}{c}
      -\nabla f_0(\xk)\\ -f(\xk)
    \end{array}
    \right],
\end{displaymath}
where we have grouped the third-order derivatives under the name
$H_3$ because we intend to neglect them.  Consider also replacing 
$\nabla^2\Lag(\xk,\lk)$ by $\nabla^2\Lag(\xk,\lkp)$. We then obtain an
approximation of the necessary optimality conditions which sits
between a first and a second-order expansion and is obtained by
solving 
\begin{equation}\label{pgm:nlpqqp}
  \begin{array}{lrlc}
    & \min &f_0(\xk)+\nabla{f_0}(x^{(k)})^T d +
    \half d^T\nabla^2{f_0}(x^{(k)})d\\ &\mbox{s.t.}
    &f_i(x^{(k)})+\nabla f_i(x^{(k)})^T d + \half d^T \nabla^2
    f_i(x^{(k)}) d \le 0, 1 \le i \le m\\
    &&d^Td \le \delta^2,
  \end{array}
\end{equation}
without the additional trust-region, which is added to ensure a bounded
solution.

Such a straightforward subproblem has often been considered, but has,
just as often, been discarded as unsolvable.  One notable exception is
an algorithm by Maany \cite{Maa:85} developed, interestingly enough,
because the standard \sqp\ approach failed on the highly nonlinear
orbital trajectory problems they were studying.  (See \cite{DHM:84}.)
Because (\ref{pgm:nlpqqp}) is a closer approximation to the original problem
(\ref{pgm:nlp}) than the quadratic program, we expect it to be a better
subproblem to solve in a sequential programming approach and, in fact
we have the following,
\begin{lemma} \label{lem:nlpqqp_foc}
  Assume that $\xk$ is feasible for (\ref{pgm:nlp}).  If the (\ref{pgm:nlpqqp}) subproblem
  is solved by $d=0$ with multipliers $\lambda$, then the pair of
  vectors $\xk$ and $\lambda$ satisfies the first-order conditions and
  second-order conditions of (\ref{pgm:nlp}).  Conversely, if $\xk$ and $\lambda$
  satisfy the first and second-order necessary conditions of (\ref{pgm:nlp}),
  then the pair of vectors $d=0$, $\lambda$ satisfy the first and
  second-order conditions of (\ref{pgm:nlpqqp}).
\end{lemma}
This implies that the (\ref{pgm:qqp}) subproblem does better than the (\ref{pgm:qp}) 
subproblem since they both solve the first-order conditions but only
the former guarantees second-order optimality conditions.  This is
expected of a trust-region approach.

Is also does better by providing second-order multiplier estimates in
the sense that the multipliers $\lkp$ obtained from (\ref{pgm:nlpqqp}) satisfy 
\begin{displaymath}
  \minpgms{\|\nabla f_0(\xk)+\nabla^2\Lag(\xk,\eta)d+
    \sum \eta_i\nabla f_i(\xk)\|^2_2}{\eta\in\Rm}.
\end{displaymath}
If we are close to the solution we therefore obtain, directly from the
solution of the subproblem, not only a good search direction in primal
space, but better multiplier estimates than provided by the standard
(\ref{pgm:qp}) subproblem. (For more details on second-order multiplier
estimates, see 

\subsubsection{Quadratically Constrained Programming}
\label{sect:q2p}
Note that, for simplicity, we assume that our constraints are
nonlinear.  Linear constraints have to be treated differently,
essentially squared, see \cite{PoReWo:94}. Equivalently, linear
constraints may be eliminated or mapped to a linear constraint in
matrix space.

Homogenization of (\ref{pgm:nlpqqp}), obtained by adding a component $d_0$ to the
vector $d$, together with the constraint $d_0^2=1$, yield the
semidefinite relaxation,
\begin{equation}\label{pgm:psdp}
  \minpgms{\inner{P_0}{Y}}{ \inner{E_{0}}{Y}=1,
  \inner{P_i}{Y} \le 0 , 1 \le i \le m, \inner{P_I}{Y}\le
  \delta^2},
\end{equation}
where 
\begin{displaymath}
  P_i = \left[ \begin{array}{ccc}-a_i & \nabla f_i(\xk)^T &0 \\ \nabla
      f_i(\xk) & \nabla^2 f_i(\xk) & 0\\
      0&0&0
    \end{array} \right],~
  a_i = -2f_i(\xk), ~~0 \le i \le m,
\end{displaymath}
and where $E_{0}$ and $P_I$ have their usual definitions,
\begin{displaymath}
  E_{0} = \left[\begin{array}{cc}1&0\\0&0\end{array}\right],
  P_I=\left[\begin{array}{cc}0&0\\0&I\end{array}\right],
\end{displaymath}
and $Y \succeq 0$.

But this relaxation may be infeasible if the current
estimate is too far from the feasible region.  To overcome this
difficulty in \sqp, Vardi suggested a heuristic shift of
the linear constraints.  We do a related shift of our second-order
constraints by allowing the additional component $d_0$ to take values
between zero and one.  That is, we change $d_0^2=1$ to $d_0^2 \leq 1$.
This additional relaxation allows for a feasible subproblem. Of course
we would want $d_0$ to be as close to 1 as possible and examination of
the subproblem shows that it automatically tries to make $d_0$
``large''.  We need no heuristic to choose a Vardi-type parameter.

The dual program to (\ref{pgm:psdp})\ is then 
\begin{equation}\label{pgm:dsdp}
  \maxpgms{-\lambda_0 }{ P_0+\lambda_0 E_{0}+\sum_{i=1}^m
  \lambda_iP_i+\lambda_IP_I \succeq 0, \lambda \in \RE \times \RE^m_{+}}.
\end{equation}

Solving the above primal-dual pair (\ref{pgm:psdp}),(\ref{pgm:dsdp}), in the case of
gap-free (\ref{pgm:nlp}), is enough since, as we have seen, the first column is
optimal for the quadratic approximation. But, in general, we need an
appropriate merit function to ensure sufficient decrease at each step
and guarantee global convergence of the algorithm, whether we use a
line search or a trust-region strategy.

After solving the (\ref{pgm:qqp})\ subproblem for a direction $d\ne{0}$, the next
iterate is obtained by $x^{(k+1)}=\xk+d$. This new point serves for
the expansion of a new problem by second-order polynomials and we
iterate until the subproblem yields $d=0$.  As with any trust-region
based algorithm, we adjust the trust-region radius according to the
ratio of predicted improvement to actual improvement.  At the end, we
have a solution satisfying both first and second-order conditions of
(\ref{pgm:nlp}).  Somewhat more formally, Algorithm sqqp???
describe the approach.



\end{slide}
\begin{slide}{}

  {\bf Sequential Quadratically Constrained Programming}
\begin{enumerate}
 \item
  \begin{enumerate}
    \item[$\bullet$] Given $f_i,\nabla f_i,\nabla^2
    f_i,x^{(0)}$\hfill {\em Functions and derivatives}
    \item[$\bullet$] Given $\epsilon$\hfill {\em Steplength tolerance}
    \item[$\bullet$] $k:=0$\hfill {\em Iteration count}\\
    {\bf REPEAT}\\
    \item[$\bullet$] $Y\in\mbox{argmin}\{\inner{P_0}{Y}:\inner{P_i}{Y}\le
    0,\inner{E_{0}}{Y}=1,Y \succeq 0 \};$ 
        -------       \hfill {\em Solve semidefinite pair}
    \item[$\bullet$] $\lkp\in\mbox{argmax}\{-\lambda_0:
    P_0+\sum\lambda P_i+\lambda_0 E_{0}\succeq 0,\lambda \in \RE \times \RE^m_{+}\};$
    \item[$\bullet$] $d := P_R(Y);$\hfill {\em Project down by first column}
    \item[$\bullet$] $\xkp := \xk + d;$ \hfill {\em New point}
    \item[$\bullet$] $r^k :=
    \frac{\varphi(\xk)-\varphi(\xkp)}{q_0(\xk)-q_0(\xkp)} 
                        $\hfill {\em Decrease ratio}\\
    {\bf IF }{$(r^{k}<\frac 14)$}
    \item[$\bullet$] $\delta=\delta/4$\hfill {\em Bad model, shrink trust-region}\\
    {\bf ELSE IF }{$(r^{k}> \frac 34)$ and $\|\xkp-\xk\|=\delta$}
    \item[$\bullet$] $\delta=2\delta$\hfill {\em Good model, expand trust-region}\\
    {\bf ENDIF ENDIF }
    \item[$\bullet$] $k:=k+1$\hfill {\em Bump iteration}\\
    {\bf UNTIL }{$(\|d\| \le \epsilon)$}\hfill {\em Attained optimality}
    \item[$\bullet$] Find maximal $d\in{\cal N}(\nabla^2 \Lag)$ such that $f(\xk+d)\le 0$
    \item[$\bullet$] $\xk:=\xk+d;$\hfill {\em Nullspace move}
    \item[$\bullet$] {\bf return}($ \xk, \lk$)\hfill {\em Primal iterate and multiplier}
  \end{enumerate}
\end{enumerate}



If the (\ref{pgm:nlpqqp}) subproblem is convex, or more generally, if it is an
instance without duality gaps, then solving the semidefinite
relaxation, which may be done efficiently, will be enough since
the primal semidefinite solution will be rank one.  We will have a
pair of primal-dual vectors satisfying the sufficient conditions for
optimality of (\ref{pgm:qqp}).

This takes care of the convex case and of many non-convex cases.  In
other cases, we move along the nullspace of the Lagrangean until
we hit one of the constraints.  This nullspace-restricted step
improves the objective value even if it does not lead to an optimal
solution.

\subsubsection{Conclusion}
Efficient approaches to unconstrained optimization based on Newton's
method all involve local quadratic models of the objective function.
Yet for constrained optimization, the extension of Newton's
method, \sqp, uses linear approximations.  Some second-order
information is included in the model, but in an aggregate form. 

In this chapter we have outlined an approach that deals more closely
with the true quadratic model of the problem at hand.  One of the key
features is the relationship between the Lagrangean and Semidefinite
relaxations which leads to what we have called the \sqqp\ algorithm
for general nonlinear programs. 

This algorithm builds second-order approximations of both the
objective function and the constraints and then solves the Lagrangean
relaxation of this quadratic model via semidefinite programming. The
approach provides a stronger relaxation than the standard quadratic
program used in \sqp\ methods; at every step it provides better
multiplier estimates; it handles potential infeasibility of the
subproblem in a straight-forward manner; finally, it aims at solutions
satisfying both first and second-order optimality conditions.  Many
implementation issues still need to be resolved but the recent
advances in numerical solutions of large semidefinite programs encourage
further study. 

As a final note, we should make clear that it may turn out that the
semidefinite relaxation is not exactly the right one to use for
efficiency reasons. But the point remains that we are nowadays in a
position to do better than the linear relaxations so popular during
the seventies and eighties.  Because we had at our disposals good
linear programming solvers, the world seemed linear or, at least,
mathematical models tried to make it so.  We now have good solvers for
quadratic programs, either via semidefinite relaxations or, possibly
via some second-order cone relaxation, and we should make full use of
these new tools.


\end{slide}
\begin{slide}{}


\subsection{Matrix Completions}
\label{sect:matrixcompl}
\subsubsection{Outline}
\begin{enumerate}

\item[$\bullet$]
background
\item[$\bullet$]
Positive Semidefinite matrix completions
\item[$\bullet$]
Euclidean Distance matrix completions
\item[$\bullet$]
exploiting sparsity
\item[$\bullet$]
A New Characterization of EDMs using Gale Matrices

\end{enumerate}
\end{slide}
\begin{slide}{}



\subsubsection{Positive Definite Completions
of Partial Hermitian Matrices}
\label{sect:psdcomplpartial}
\begin{itemize}
\item[$\bullet$]
   $\GG(V,E)$ {\em finite undirected graph}, $\qquad$ (node set $V$,
           edge set $E$)\\
\item[$\bullet$]
    $A(\GG)$ is a {\em $\GG$-partial matrix},  $\qquad$
          ($a_{ij}$ defined {\em iff} $\{i,j\} \in E$)\\
\item[$\bullet$]
    $A(\GG)$ is a {\em $\GG$-partial positive matrix} if
           $a_{ij}=\overline{a_{ji}}, \forall \{i,j\} \in E$ and all existing
           principal minors are positive.\\
\item[$\bullet$]
    with ${\mathcal J}=(V,\bar{E}), E \subset \bar{E}$ {\em a 
    $\mathcal J$-partial matrix
    $B({\mathcal J})$ extends} the $\GG$-partial matrix $A(\GG)$ if
    $b_{ij}=a_{ij}, \forall \{i,j\} \in E$\\
\item[$\bullet$]
    $\GG$ is {\em positive
    completable} if every $\GG$-partial positive matrix can be
    extended to a positive definite matrix.
\end{itemize}
\end{slide}
\begin{slide}{}
\begin{center}
$\GG$ is {\bf chordal} if there are no minimal cycles of length $\geq
4$.\\
    (i.e. every cycle of length $\geq 4$ has a chord)
\end{center}
~~\\
\begin{thm}
(Grone, Johnson, Sa, Wolkowicz)\\
$\GG$ is positive completable {\bf iff} $\GG$ is chordal.
\epr
\end{thm}
~~\\
Equivalently, positive completability can be
expressed as strict feasibility in SDP:
\[\begin{array}{cl}
\trace E_{ij}P=a_{ij}, & \forall \{i,j\} \in E\\
P \succ 0
\end{array}\]
where $E_{ij}=e_ie_j^T+e_je_k^T$
\label{endsect:psdcomplpartial}
\end{slide}
\begin{slide}{}

\subsubsection{Approximate Positive Semidefinite Completions}
\label{sect:approxpsdcompl}

GIVEN:
\begin{itemize}
\item[$\bullet$]
 $H=H^T \geq 0$ a real, nonnegative (elementwise) {\bf symmetric
matrix of weights} \\
 (with positive diagonal elements $ H_{ii} > 0,~ \forall i$)

\item[$\bullet$]
$A=A^*$, the
 {\bf given partial Hermitian matrix}\\
i.e. some elements approximately fixed,
others free (for notational purposes, assume
free elements set to 0 if not specified.)
\end{itemize}

\end{slide}
\begin{slide}{}


$||A||_F = \sqrt{ \tr A^*A}$
{\em Frobenius norm},
$\circ$ denotes {\em Hadamard product}.\\
\[ \begin{array}{cc}
  f(P):=||H \circ (A-P) ||_F^2 
\end{array}
\]
~~\\
~~\\
{\bf the weighted, best approximate, completion problem}
\[
\mbox{(AC)} \qquad
\begin{array}{ccc}
       \mu^*:=&\min &f(P) \\
 &  \mbox{~subject to~} & \KK P=b\\
  &  &  P \succeq 0,
    \end{array}
\]
where $\KK: \hn \rightarrow {\cal C}^m$ linear operator




\end{slide}
\begin{slide}{}

\[
\mbox{\bf Lagrangian}: \qquad
          L(P,y,\Lambda) = f(P)  + \left<y,b-\KK P\right> - \tr \Lambda P
\]
{\bf Dual problem:} 
 \[
\mbox{ (DAC)} \qquad
\begin{array}{ccc}
       \max &f(P) +\left<y,b-\KK P\right>- \tr \Lambda P \\
   \mbox{~subject to~} &  \nabla f(P)  -\KK^*y- \Lambda =  0\\
          &       \Lambda \succeq 0.
    \end{array}
\]
\begin{thm}  \label{thm:optcond}
The matrix $\bar{P}\succeq 0$ and vector-matrix
$\bar{y},\bar{\Lambda} \succeq 0$ solve AC and DAC resp. {\bf iff}
\[
\begin{array}{cc}
 \KK\bar{P}  = b & \mbox{primal feas.}\\
 2H^{(2)} \circ (\bar{P}-A)-\KK^*\bar{y}
  -\bar{\Lambda} =  0 & \mbox{dual feas.}\\
\tr \bar{\Lambda} \bar{P} = 0 & \mbox{compl. slack.}
\end{array}
\]
\epr
\end{thm}


\end{slide}
\begin{slide}{}
For simplicity, replace  $\KK$ with appropriate weights in $H$.\\
Use {\em square} perturbed optimality conditions.\\
(For $P,\Lambda \succeq 0$: $\trace P\Lambda = 0 \iff P\Lambda = 0$.)

\[
\begin{array}{cc}
 2H^{(2)} \circ (P-A) -\Lambda =  0 & \mbox{dual
feasibility}\\
-P + \mu \Lambda^{-1}  = 0 & \mbox{perturbed C.S.}\\
\end{array}
\]

Linearize second equation and solve for $h$ and $l$

\[
h= \mu \Lambda^{-1} - \mu \Lambda^{-1} l \Lambda^{-1}-P
\]
\[
l= \frac 1{\mu}\left\{
     - \Lambda (P+h) \Lambda
 \right\}+\Lambda
\]



\end{slide}
\begin{slide}{}

{\bf Dual Step First:} (if many elements of $P$ are free; similar
approach for primal step first if many elements are fixed)

We can eliminate the primal step $h$ and solve for the dual
step $l$.
\[
\begin{array}{ccl}
l & = &  2 H^{(2)} \circ h + ( 2H^{(2)} \circ (P-A)
-\Lambda)\\
  & = &  2 H^{(2)} \circ  (\mu \Lambda^{-1} - \mu \Lambda^{-1}\\
&& ~~~~~~~~ l \Lambda^{-1}  -P)+ ( 2H^{(2)} \circ (P-A)
-\Lambda).
\end{array}
\]
Equivalently, we get the Newton equation
\[
\begin{array}{ccl}
 2 H^{(2)} \circ (\mu \Lambda^{-1} l \Lambda^{-1}  ) + l =
 2 H^{(2)} \circ (\mu \Lambda^{-1} -A  ) -\Lambda.
\end{array}
\]

This shows that $l,\Lambda$ have the same sparsity pattern as
$H$\\
(order is number of nonzeros in $H$/2.)


\end{slide}
\begin{slide}{}


{\bf Each test appears on one line and includes 20 test problems.}\\

~~\\
\begin{tiny}
\begin{flushleft}
\begin{tabular}{|c|c|c|c|c|c|c|c|c|}\hline
 dim &  toler  &$H$dens./infty&$A$psd&cond(A)&$H$pd&min/max&iters\\ 
\hline
  15 &   $10^{-5}$& .0751/.02  & yes & 222.5& no & 8/17 & 10.3  \\
  15 &   $10^{-6}$& .1/.95  & yes & 19.6 & no & 10/23 & 15.6 \\
  15 &   $10^{-6}$& .01/.95  & yes & 21. & no & 10/20 & 13.2 \\
  19 &   $10^{-6}$& .005/.1  & yes & 14.8 & no & 10/18 & 12.9 \\
  21 &   $10^{-6}$& .005/.1  & yes & 20.3 & no & 8/24 & 15 \\
  38 &   $10^{-6}$&    1/.99  & yes & 49.6 & yes & 14/24 & 16. \\
  45 &   $10^{-6}$&    1/.99  & yes & 46.8 & yes & 15/22 & 17. \\
  55 &   $10^{-6}$&    1/.99  & yes & 37.2 & yes & 15/30 & 17.5 \\
  85 &   $10^{-5}$& .0219/.02  & yes & 1374.5& no & 16/23 & 18.9  \\
  95 &   $10^{-5}$& .0206/.02  & yes & 2.7 & no & 8/14 & 11.1  \\
  95 &   $10^{-6}$& 1/.999 & yes & 196. & yes & 14/18 & 16.8  \\
 145 &   $10^{-6}$& .01/.997 & yes &  658.5 & yes & 13/17 & 14.9 \\
\hline
\end{tabular}
~~\\
data for primal-step-first (20 problems per test): \\
dimension;  tolerance for duality gap;\\
 density of nonzeros in $H$/ density of infinite values in $H$;\\
positive semidefiniteness of $A$; condition number of $A$; 
positive definiteness of $H$;\\
min and max number of iterations; average number of iterations.

\end{flushleft}
\end{tiny}



\end{slide}
\begin{slide}{}


\begin{flushleft}
\begin{tiny}
\begin{tabular}{|c|c|c|c|c|c|c|c|c|}\hline
 dim& toler&$H$dens./infty&
           $A$psd &cond(A)&$H$pd&min/max&iters\\ \hline
  60   & $10^{-6}$& .01/.001  & yes & 79.7& no & 15/23 & 16.8  \\
 65   & $10^{-6}$& .015/.001  & yes & 49.9& yes & 18/24 & 21.3  \\
 83   & $10^{-6}$& .007/.001  & no & 235.1 & no & 24/29 & 25.5 \\
 85   & $10^{-5}$& .008/.001  & yes & 94.7 & no & 11/17 & 13.1 \\
 85   & $10^{-6}$& .0075/.001  & no & 299.9 & no & 23/27 & 25.2 \\
 87   & $10^{-6}$& .006/.001  & yes & 74.2 & yes & 14/19 & 16.9 \\
 89   & $10^{-6}$& .006/.001  & no & 179.3 & no & 23/28 & 15.2 \\
110   & $10^{-6}$& .007/.001  & yes & 172.3& yes & 15/20 & 17.8  \\
155   & $10^{-6}$& .01/0  & yes &643.9& yes & 14/18 & 15.3  \\
655   & $10^{-6}$& .017/0  & yes &1.4& no & 14/14 & 14.  \\
755   & $10^{-6}$& .002/0  & yes &1.5& no & 15/15 & 15.  \\
\hline
\end{tabular}
~~\\
data for dual-step-first (20 problems per test): 
dimension;  tolerance for duality gap;\\
 density of nonzeros in $H$/ density of infinite values in $H$;\\
positive semidefiniteness of $A$; condition number of $A$; 
positive definiteness of $H$;\\
(only one test for: 655,755; {\em bottleneck} is generating large random probs)
\end{tiny}
\end{flushleft}
\label{endsect:approxpsdcompl}

\end{slide}
\begin{slide}{}

\subsubsection{Euclidean Distance Matrix Completion Problem}
\label{sect:eucldistcompl}


{\bf What are EDMs?}\\
\begin{itemize}
\item
A {\bf pre-distance matrix} (or dissimilarity matrix) is:\\
an $n \times n$ symmetric
matrix $D=(d_{ij})$ with nonnegative elements and zero diagonal
\item
A (squared) {\bf Euclidean distance matrix} (EDM) is:\\
a pre-distance matrix such that there
exists points $x^1,x^2,\ldots,x^n$ in $\RR^r$ such that
\[
d_{ij} = {\| x^i- x^j\|}^2, ~~~ i,j=1,2,\ldots,n.
\]
\item
The smallest value of $r$ is called {\bf the embedding dimension} of
$D$.  ($r$ is always $\leq n-1$, e.g. translate $x^1$ to origin)

\end{itemize}
\end{slide}
\begin{slide}{}
{\bf EDM problem:}\\
 Given a partial symmetric matrix $A$ with certain elements specified,
the Euclidean distance matrix completion problem
(EDMCP) consists in finding the unspecified elements
of $A$ that make $A$ a EDM. 

{\bf WHY?}\\
e.g.: \\
$\bullet$ 
The shape of an enzyme determines its chemical function. Once the
shape is known, then the proper drug can be designed.\\
~~\\
$\bullet$ 
distance geometry on molecules: Atoms are points in space with pairwise
distances; find a set of points which yield those distances.
\end{slide}
\begin{slide}{}
$\bullet$ {\bf approximate EDMCP},
let: $A$ be a pre-distance matrix,
$H$ be an $n \times n$ symmetric  matrix with nonnegative elements,
$\|A\|_F= \sqrt{
     \trace A^TA}$ denote the {\em Frobenius norm} of $A.$\\
$\bullet$ {\bf objective function}
\[ f(D) := {\| H \circ (A - D) \|}^2_F,   \]
where $\circ$ denotes {\em Hadamard product}.\\
$\bullet$ {\bf weighted, closest Euclidean distance matrix problem} is
\[
(CDM_0)
 \begin{array}{ccc}
          \mu^* := & \min   &   f(D)  \\
                  & \mbox{ subject to } & D \in {\DD}, 
  \end{array}
\]
where $\DD$ denotes the (convex) cone of EDMs.
\end{slide}
\begin{slide}{}
{\bf DISTANCE GEOMETRY}

A pre-distance matrix 
$D$ is a EDM if and only if $D$ is negative semidefinite on 
\[ M:=\left\{ x \in \Rn : x^T e = 0 \right\},
\]
where $e$ is the vector of all ones.

Define $V$ $n \times (n-1),$ full column rank such that $V^Te=0.$
Then
\[ \label{eq:Vmp}
J := V V^{\dagger}= I- \frac{e e^T}{n}
\]
is the orthogonal projection onto $M$,
where $V^{\dagger}$ denotes {\em Moore-Penrose generalized inverse}.

\end{slide}
\begin{slide}{}
Define the {\bf centered} and {\bf hollow} subspaces
\[ \begin{array}{rcl}
\Sc &:=&  \{ B \in \Sn :  Be = 0 \}, \\ 
\Sh& := & \{ D \in \Sn :  \diag(D) = 0 \}. 
\end{array}
\] 
Define the two linear operators
\[ \begin{array}{rcl} \label{KK} 
\KK(B)& := &  \mbox{diag}(B)\,e^T + e \, \mbox{diag}(B)^T - 2B,
\end{array} \] 
\[ \begin{array}{rcl} \label{T} 
\TT(D)& := &  -\frac 12 JDJ.
\end{array} \]
The operator $- 2 \TT$ is an orthogonal projection onto $\Sc.$ 
\begin{thm}
\begin{eqnarray*}
\KK (  \Sc) &=& \Sh, \\
\TT (  \Sh) &=& \Sc, 
\end{eqnarray*} 
and $\KK_{|\Sc}$ and $\TT_{|\Sh}$ are inverses of each other.
\epr
\end{thm}
\end{slide}
\begin{slide}{}

$D$ (hollow matrix) is EDM $\iff$ $B=\TT(D) \succeq 0$ 

$D$ is EDM $\iff$ $D=\KK(B),$ for some $B$ with $Be=0$ and $B \succeq 0$.   

In this case the embedding dimension $r$ is given
by the rank of $B$. Moreover if $B=XX^T$, then
the coordinates of the points  $x^1,x^2,\ldots,x^n$ that generate $D$ are
given by the rows of $X$ and, since $Be=0,$
it follows that the origin coincides with the  centroid
of these points.
\end{slide}
\begin{slide}{}

The cone of EDMs, $\DD$, has empty interior.\\
 This can cause problems for interior-point methods.

\[ V \cdot V : {\cal S}_{n-1}  \rightarrow {\cal S}_{n}  
\]
\[ V \cdot V : {\cal P}_{n-1}  \rightarrow {\cal P}_{n}  
\]

Define the composite operators
\[ \begin{array}{rcl} \label{KV} 
\KK_V(X)& := &  \KK( V X V^T),
\end{array} \] 
and
\[ \begin{array}{rcl} \label{TV} 
\TT_V(D)& := &  V^{\dagger}\TT( D)(V^{\dagger})^T= 
                     - \frac 12 V^{\dagger} D (V^{\dagger})^T.
\end{array} \] 

\end{slide}
\begin{slide}{}


{\bf LEMMA}
\begin{eqnarray*} 
\KK_V ( \snn) =\Sh, \\
\TT_V ( \Sh) =\snn, 
\end{eqnarray*} 
and $\KK_V$ and $\TT_V$ are inverses of each other on these two spaces.
\epr

{\bf COROLLARY}
\begin{eqnarray*} 
\KK_V(\p)  & = &  \DD , \\
\TT_V(\E) & = &\p.
\end{eqnarray*} 
\epr
\end{slide}
\begin{slide}{}

\begin{center}
Summary
\end{center}

(Re)Define the closest EDM problem:
\begin{eqnarray*}
 f(X) := {\| H \circ (A - \KK_V ( X)) \|}^2_F\\
          = {\| H \circ \KK_V(B - X) \|}^2_F, 
\end{eqnarray*}
where $B = \TT_V(A)$.\\
($\KK_V$ and $\TT_V$ are both linear operators)

{\bf
\[
(CDM)
 \begin{tabular}{ccc}
          $\mu^*$ := & $\min$   &   $f(X)$  \\
                  &  subject to  & $\A X=b $ \\ 
               &   &  $X \succeq 0.$
  \end{tabular}
\]
}

~\\

~\\
The additional constraint
using $\A : \snn \longrightarrow \Rm$, 
could represent some of the fixed
elements in the given matrix $A$.
\end{slide}
\begin{slide}{}

{\bf Primal-Dual Interior-Point Framework:}

\begin{enumerate}
\item
derive a dual program

\item
state optimality conditions for log-barrier problem (perturbed
primal-dual optimality conditions)

\item
find a search direction for solving the perturbed optimality conditions

\item
take a step and backtrack to stay strictly feasible (positive definite)

\item
Update and go to Step 3
(adaptive update of log-barrier parameter)
\end{enumerate}

\end{slide}
\begin{slide}{}

{\bf Step 1.  derive a dual program:}

$\Lambda \in \snn, \Lambda \succeq 0$ and $y \in R^m$, \\
Lagrangian is
\[
L(X,y,\Lambda) = f(X) + \langle y, b - \A(X) \rangle - 
   \langle \Lambda, X  \rangle
\]

primal program (CDM) is
\[
       =  \min_{X} \max_{\stackrel{y}{\Lambda 
                    \succeq 0}}  L(X,y,\Lambda).   
\]

dual program is:
\[
         = \max_{\stackrel{y}{\Lambda \succeq 0}} \min_{X}  
          L(X,y,\Lambda),  
\]
\end{slide}
\begin{slide}{}

The inner minimization of the convex, in $X$, Lagrangian is unconstrained
so we add the hidden constraint which makes the minimization redundant.

dual program (DCDM)
\[
 \max_{\stackrel{\nabla f(X) - \A^*y=\Lambda}{\Lambda \succeq 0}}
                f(X) + \langle y, b - \A(X) \rangle - \mbox{trace} \Lambda X.
\]
or
\[
\begin{array}{cccc} 
 & & \max & f(X)+\langle y, b - \A(X) \rangle -
      \trace  \Lambda X \\  
       &           & \mbox{subject to} & \nabla f(X) - \A^*y- \Lambda = 0  \\
       &           &             &   \Lambda \succeq 0, (X \succeq 0).   
\end{array}  \]


\end{slide}
\begin{slide}{}

the duality gap,\\
 $  f(X)-\left(f(X)+\langle y, b - \A(X) \rangle - \trace  \Lambda X
           \right),$\\
in the case of primal and dual
feasibility, is given by the complementary slackness condition:
\[
\mbox{ trace } X (\KK^*_V( H^{(2)} \circ \KK_V( {X}-B))- \A^*{y}) = 0,
\]
or equivalently 
\[
 X (\KK^*_V( H^{(2)} \circ \KK_V( {X}-B))- \A^* {y}) = 0 ,
\]
where $H^{(2)}=H \circ H.$
\end{slide}
\begin{slide}{}


\begin{thm}
Suppose that Slater's condition holds for CDM. Then
$\bar{X} \succeq 0$, and $\bar{y}$, $\bar{ \Lambda} 
      \succeq 0$ solve (CDM) and (DCDM),
respectively, if and only if the following three equations hold.
\[ \begin{array}{cc}
 \A (\bar{X}) = b  & \mbox{prim. feas.}  \\
  2\KK^*_V( H^{(2)} \circ \KK_V( \bar{X}-B)) - \A^* \bar{y} - 
                          \bar{\Lambda} =0 &
                                           \mbox{dual feas.}  \\
\trace \bar{ \Lambda} \bar{X}=0 & \mbox{C.S.}
\end{array} \]  
\epr
\end{thm}


\end{slide}
\begin{slide}{}

\begin{lem}
Let $H$ be an $n \times n$ symmetric matrix with nonnegative elements
and 0 diagonal such that the graph of $H$ is connected.  Then 
\[ \KK_V^*(H^{(2)} \circ \KK_V(I)) \succ 0 ,  \]
where $I \in \snn$ is the identity matrix.
\epr
\end{lem}

i.e. when ${\cal A}=0,$ we have a Slater point for the dual (and
primal)
\end{slide}
\begin{slide}{}

{\bf Step 2.  
state optimality conditions for log-barrier problem (perturbed
primal-dual optimality conditions):}


 The log-barrier problem for (CDM) is
\[ \min_{X \succ 0} B_{\mu}(X) := f(X) - \mu \log \det(X),  \]
where $\mu \downarrow 0$. 

For each $\mu > 0$ we take one Newton step
for solving the stationarity condition
\[
\nabla B_{\mu}(X) = 2 \KK^*_V(H^{(2)} \circ \KK_V(X - B)) - \mu
X^{-1}=0.
\]

Let
\[
C := 2 \KK^*_V(H^{(2)} \circ \KK_V( B))
= 2 \KK^*_V(H^{(2)} \circ A).
\]

Then the stationarity condition is equivalent to
\[
\nabla B_{\mu}(X) = 2 \KK^*_V\left(H^{(2)} \circ \KK_V(X)\right)
            - C - \mu X^{-1}=0.
\]



\end{slide}
\begin{slide}{}

equating $\Lambda = \mu X^{-1}$ and multiplying through by $X$

optimality conditions,
$F:=\left( \begin{array}{c} F_d \\ F_c  \end{array} \right)=0,$
\[ \begin{array}{llccl}
 2 \KK^*_V\left(H^{(2)} \circ \KK_V(X)\right) - C
- \Lambda &=&0 &  \mbox{dual feas.} \\
 \Lambda X - \mu I&=&0  &  \mbox{pert. C.S.},
\end{array} \]
(an OVERDETERMINED nonlinear system since $\Lambda X$ not
symmetric)

estimate of the barrier parameter
\[  \mu = \frac{1}{n-1} \mbox{ trace }\Lambda X    \]


\end{slide}
\begin{slide}{}

$\sigma_k$ centering parameter \\
${\cal F}^0$ set of strictly feasible primal-dual
points\\
 $F^\prime$ derivative of $F$

\begin{alg}
 (p-d i-p framework:)\\
{\bf Given} $(X^0,\Lambda^0) \in {\cal F}^0$\\
{\bf for} $k=0,1,2 \ldots $\\
\hspace{.5in}{\bf solve} for the search direction\\
     \[
  ~~~~F^{\prime}(X^k,\Lambda^k)
\left(
\begin{array}{ccc}
\delta X^k \\  \delta \Lambda^k
\end{array}
\right)
= 
\left(
\begin{array}{ccc}
-F_d \\ -\Lambda^k X^k+ \sigma_k \mu_k I
\end{array}
\right)
\]
\hspace{.5in}where $\sigma_k$ centering,
 $\mu_k=\frac {\trace X^k\Lambda^k}{(n-1)}$

\[ (X^{k+1},\Lambda^{k+1}) = (X^k,\Lambda^k) +
\alpha_k
(\delta X^k,    \delta \Lambda^k)
\]
\hspace{.5in}so that $(X^{k+1},\Lambda^{k+1}) \succ 0$\\
{\bf end (for)}.
\end{alg}




\end{slide}
\begin{slide}{}
search direction (Gauss-Newton direction, cannot use {\em square}
system because of $\KK_V$) - the Frobenius norm lss of
$ F^{\prime} s = - F$, i.e.
\[ \begin{array}{rcll}
 2 \KK^*_V\left(H^{(2)} \circ \KK_V (h)\right) - l&=&-F_d    \\ 
  \Lambda h + l X&=&-F_c.
\end{array} \]
$t(n)=\frac {(n+1)n}2$ be dimension of $\Sn.$
\[
F^{\prime} s=
\left[  \begin{array}{cc}
F^{\prime}_{u1}& F^{\prime}_{u2}\\
F^{\prime}_{l1} & F^{\prime}_{l2}
\end{array} \right]
\left(  \begin{array}{cc}
h \\
l
\end{array} \right)
=
rhs=
\left(  \begin{array}{cc}
rhs_1 \\
rhs_2
\end{array} \right).
\]
Operator $F^{\prime}$ maps $\RR^{2(t(n-1))}$ to $\RR^{t(n-1)+(n-1)^2}.$ 

%\end{slide}
%\begin{slide}{}
%
%
%%    \begin{figure}[htb]
%%     \centering
%%     \centerline{\
%%\psfig{figure=fig11513.ps,width=5.3in}
%%      \label{fig2}
%%   \caption{Approximate Completion Problem}
%%    \end{figure}
%
%
%
%\begin{center}
%\psfig{figure=fig11513.ps,width=3.1in}
%\end{center}
%\end{slide}
%\begin{slide}{}
%
%
%\begin{center}
%\psfig{figure=fig13513.ps,width=3.1in}
%\end{center}
%\end{slide}
%\begin{slide}{}
%
%\begin{center}
%\psfig{figure=fig8513.ps,width=3.1in}
%\end{center}
%\end{slide}
%\begin{slide}{}
%
%
%\begin{center}
%\psfig{figure=fig9513.ps,width=3.1in}
%\end{center}
%
%
\label{endsect:eucldistcompl}

\end{slide}
\begin{slide}{}

\subsubsection{Large Sparse Problems}
Instead of projecting and reducing the dimension to get
Slater's condition, add a variable and increase the dimension.

\begin{lem}
\label{lem:newcharacthollow}
Let
\[
\begin{array}{lcl}
\DD &:=& \{X \in \Sh : v^Te=0 \quad \Rightarrow \quad v^TXv
\leq 0
           \},\\
\DD_{-1} &:=& \left\{X \in \Sh : 0=\max\{ v^TXv : v^TEv=0  \} \right\},\\
\DD_0 &:=&\{ X \in \Sh :
       X - \alpha ee^T \preceq 0, \quad \mbox{for some } \alpha
        \},\\
\DD_1 &:=&\{ X \in \Sh :
X - \alpha ee^T \preceq 0, \quad \forall ~ \alpha \geq
\bar{\alpha},
            \mbox{ for some } \bar{\alpha} \}.
\end{array}
\]
Then
\beq  \label{eq:subsets}
  {\rm ri} \left(\DD \right) \subset
  \DD_0  = \DD_1 \subset \DD \subset \DD_{-1} \subset \overline{\DD_0}.
\eeq
\end{lem}


\end{slide}
\begin{slide}{}


\bpr
That $\DD =\DD_{-1}$ is clear and provides motivation for the other
inclusions based on Lagrange multipliers.
Suppose $\bar{X} \in {\rm ri} \left(\DD \right)$ (i.e.
$v^Te=0, v \neq 0 \Rightarrow v^T\bar{X}v < 0$) but
$\bar{X} \notin  \DD_0$.
Then, for each $\alpha \geq 0$,
there exists $w_{\alpha}$ with $||w_{\alpha}||=1$, such that
$w_{\alpha} \rightarrow \bar{w}$, as $\alpha \rightarrow
\infty$ and
\[
w_{\alpha}^T(\bar{X} - \alpha  ee^T)w_{\alpha} > 0, \quad
\forall ~
            \alpha\geq 0, \qquad \mbox{i.e.}
\]
\[
w_{\alpha}^T\bar{X}w_{\alpha} >
\alpha w_{\alpha}^T  ee^Tw_{\alpha} , \quad \forall ~
            \alpha\geq 0.
\]
Since $w_{\alpha}$ converges and the left-hand-side of the
above
inequality must be finite, this implies that $e^T \bar{w} =
\bar{w}^T\bar{X}\bar{w} = 0$, a contradiction. Therefore,
  ${\rm ri} \left(\DD \right) \subset \DD_0$.
That $\DD_0 = \DD_1$ is clear.

Now suppose that $\bar{X} - \alpha ee^T \preceq 0, ~ \alpha
\geq 0$. Let $v^T
e
= 0$. Then $0 \geq v^T(\bar{X} - \alpha ee^T)v=v^T\bar{X}v$,
i.e.
$\DD_0 \subset \DD$. The final inclusion comes from the first
and
the fact that $\DD$ is closed.
\epr


%\end{slide}
%\begin{slide}{}
%
%\begin{cor}
%\label{cor:newcharacthollow}
%Let
%\[
%\begin{array}{rcl}
%\EE &:=& \{X \in \Sh : v^Te=0 \quad \Rightarrow \quad v^TXv
%\leq 0
%           \},\\
%\EE_0 &:=&\{ X \in \Sh :
%       X - \alpha ee^T \preceq 0, \quad \mbox{for some } \alpha
%        \},\\
%\EE_1 &:=&\{ X \in \Sh :
%X - \alpha ee^T \preceq 0, \quad \forall ~ \alpha \geq
%\bar{\alpha},\\
%&& ~~~~~~~~~~~~ \mbox{ for some } \bar{\alpha} \}.
%\end{array}
%\]
%Then
%\beq  \label{eq:subsetsE}
%  \EE = \EE_0  = \EE_1.
%\eeq
%\end{cor}
%\bpr
%The proof is similar to that in the above Lemma
%\ref{lem:newcharact}.
%We only include the details about the closure.
%
%Suppose that $0 \neq X_k \in \EE_0$, i.e.
%$\diag(X_k)=0, X_k \preceq \alpha_k E$, for some $\alpha_k$;
%and, suppose that
%$X_k \rightarrow \bar{X}$. Since $X_k$ is hollow it has exactly
%one positive eigenvalue and this must be smaller than
%$\alpha_k$.
%However, since $X_k$ converges to $\bar{X}$, we conclude that
%$\bar{X} \leq \lambda_{\max}(\bar{X}) E$, where
%$\lambda_{\max}(\bar{X})$ is the largest eigenvalue of
%$\bar{X}$.
%\epr
%


\end{slide}
\begin{slide}{}

let: $E=ee^T$;
$f(P) := {\| H \circ (A - P) \|}^2_F$;\\
$\KK$ lin. operator with constraint $\diag (P)=0$.\\
{\bf primal problem} is:
\[
({\rm CDM})
 \begin{tabular}{ccc}
          $\mu^*$ := & $\min$   &   $f(P)$  \\
                  &  subject to  & $\KK P=b $ \\
               &   &  $\alpha E-P \succeq 0$
  \end{tabular}
\]
and {\bf dual problem (DCDM)} is
\[
\begin{array}{rcc}
   \mu^*=\nu^*:= \max &f(P) +\left<y,b-KP\right>- \tr \Lambda (\alpha
E-P)\\
 ~~~  \mbox{s.t.} &  \nabla_P f(P)  -\KK^*y+ \Lambda =
0\\
 ~~~                      & -\trace \Lambda E =  0\\
          ~~~&       \Lambda \succeq 0.
    \end{array}
\]

(Slater's holds for primal but fails for dual.)

\end{slide}
\begin{slide}{}

(perturbed) Optimality Conditions are:
\[
\begin{array}{cl}
  \diag (P) = 0  & \mbox{primal feas.}\\
 2H^{(2)} \circ (P-A)-\Diag(y) +\Lambda =  0, \\
  ~~~  -\trace \Lambda E= 0  & \mbox{dual feas.}\\
-(\alpha E -P) + \mu \Lambda^{-1}  = 0, ~
                  & \mbox{pert. C.S.}\\
\end{array}
\]



\end{slide}
\begin{slide}{}

 \[  \begin{array}{rcl}
h & \mbox{denotes the step for}&  P\\
w & \mbox{denotes the step for}&  \alpha\\
l & \mbox{denotes the step for}&  \Lambda\\
s & \mbox{denotes the step for}&  y.
\end{array}
\]

maintain
\[
\diag(h)=\diag(P)=0
\]



\end{slide}
\begin{slide}{}

linearization of complementary slackness
\[
\begin{array}{rcl}
-(\alpha + w)E+(P+h)+ \mu \Lambda^{-1} -
                     \mu \Lambda^{-1} l \Lambda^{-1}&=&0,
\end{array}
\]


solve for $h$
\[
\begin{array}{rcl}
h &=& -\mu \Lambda^{-1} + \mu \Lambda^{-1}
           l \Lambda^{-1}-P +(\alpha  +w)E.
\end{array}
\]
(or solve for $l$)

linearization dual feasibility
\[
\begin{array}{rcl}
2 H^{(2)} \circ h -\Diag(s) +l &=& -( 2H^{(2)} \circ (P-A)\\
               && ~~~ -\Diag(y)+\Lambda)\\
-\trace  l E   &=& \trace \Lambda E
\end{array}
\]







\end{slide}
\begin{slide}{}

substitute for $h,s$\\
Newton equation is
\[
\begin{array}{rcl}
 2 H^{(2)} \circ \left(
wE +\mu \Lambda^{-1} l \Lambda^{-1}  \right) -\Diag\diag (l)+ l \\
 ~~~~~ = 2 H^{(2)} \circ \left\{\mu \Lambda^{-1} +A  -\alpha E \right\}
          +\Diag(y) -\Lambda\\
\diag\left( \mu \Lambda^{-1} l \Lambda^{-1}\right)  + we \\
  ~~~~~ = \diag\left(\mu \Lambda^{-1}\right) - \alpha e\\
\trace(lE)  = -\trace(\Lambda E).
\end{array}
\]
square system, order $1+nnz$ where $nnz$ are the number of
nonzeros in the upper triangular part of $H$, ($\diag(H)=0$).



\end{slide}
\begin{slide}{}

$F$ denotes $(nnz+n) \times 2$ matrix\\
row $p$ contains indices of the $p$-th nonzero, 
upper triangular, element of $H+I$ ordered by columns,
\[
\begin{array}{lcc}
\left\{ (F_{p1},F_{p2})_{p=1, \ldots nnz+n} \right\}\\
  ~~~~~~~~~~  = \left\{ ij : H_{ij} \neq 0, i \leq j, \mbox{ ordered
                by columns} \right\}.
\end{array}
\]
$\delta_{ij}$ is {\em Kronecker delta function}\\
 $\delta_{(ij)(kl)}$  is 1 if $(ij)=(kl)$, 0 otherwise.

$E_{ij} = \left( e_ie_j^T+e_j^Te_i\right)/\sqrt{2}$,
$ij$ unit matrix in $\Sn$, where 
$E_{ij} = \left(e_ie_j^T+e_j^Te_i\right)/2$ if $i=j$.\\
(orthonormal basis of $\Sn$)



\end{slide}
\begin{slide}{}


operator equation:\\
\[
\begin{array}{lcl}
{\bf \mbox{$k\neq l, i\neq j$ LHS}} = \\
= \tr E_{kl} \left\{ 2 H^{(2)} \circ 
\left( \mu\Lambda^{-1} E_{ij} \Lambda^{-1} \right) 
   -\Diag \diag (E_{ij}) \right.\\
   ~~~~~~~~~~~~~~~~~~ \left. + E_{ij}\right\} \\
= \mu\tr (e_ke_l^T+e_le_k^T) \left(H^{(2)} \circ \Lambda^{-1}
    (e_ie_j^T+e_je_i^T)\Lambda^{-1} \right)\\
      ~~~~~~~~~~~~~~~~ + \delta_{(ij)(kl)}\\
=  \mu   \left[
     2e_{l}^T \left(H^{(2)} \circ \Lambda^{-1}_{:,i}
                   \Lambda^{-1}_{j:} \right)e_k+
     2e_{k}^T \left(H^{(2)} \circ \Lambda^{-1}_{:,i}
                   \Lambda^{-1}_{j:} \right)e_l \right] \\
   ~~~~~~~~~~~~    + \delta_{(ij)(kl)};
\end{array}
\]
\[
\begin{array}{lcl}
{\bf \mbox{$k\neq l, i\neq j$ LHS}}=\\
=  2 \mu H^{(2)}_{kl} \left(
     \Lambda^{-1}_{li} \Lambda^{-1}_{jk}+
  \Lambda^{-1}_{ki} \Lambda^{-1}_{jl} \right)  + \delta_{(ij)(kl)};\\
~~\\
{\bf \mbox{$k\neq l, i= j$ LHS}}=\\
= \tr E_{kl} \left\{ 2\mu H^{(2)} \circ 
\left[ \Lambda^{-1} E_{jj} \Lambda^{-1} \right]
   -\Diag \diag (E_{jj})  + E_{jj}\right\} \\
= 2\sqrt{2}\mu\tr e_ke_l^T \left(H^{(2)} \circ \Lambda^{-1}
     e_je_j^T \Lambda^{-1} \right) \\
=  2\sqrt{2} \mu H^{(2)}_{kl} \left(
     \Lambda^{-1}_{lj} \Lambda^{-1}_{jk} \right);\\
{\bf \mbox{$k =  l, i\neq j$ LHS}}=\\
~~~ =  \sqrt{2}\mu \Lambda^{-1}_{ki} \Lambda^{-1}_{jk},
              \quad k= 1, \ldots n;\\
{\bf \mbox{$k =  l, i= j$ LHS}}=\\
=  \mu \Lambda^{-1}_{ki} \Lambda^{-1}_{ik},
                  \quad k= 1, \ldots n.
\end{array}
\]


\end{slide}
\begin{slide}{}


last column of LHS, matrix $l=0$ and $w=1$:
\[
\begin{array}{rcl}
{\bf \mbox{$w=1, k \neq l$ LHS}} &=&
              \tr  \left(E_{kl} ( 2H^{(2)} \circ E)\right);\\
{\bf \mbox{$w=1, k=l$ LHS}} &=& 1.
\end{array}
\]
last row of LHS:
\[
\begin{array}{rcl}
{\bf \mbox{$ i \neq j$ LHS}}
               &=& \tr  \left(E_{ij} E)\right) =   \sqrt{2};\\
{\bf \mbox{$ i = j$ LHS}}
               &=&   1.
\end{array}
\]


\end{slide}
\begin{slide}{}


Newton system is:
\[
 \sMat\left[ L (\svec (l))\right] = \sMat\left[ \svec (RHS) \right],
\]
$\svec(S)$ vector formed from 
nonzero elements of columns of
upper triangular part, where
strict upper triangular part is multiplied by
$\sqrt{2}$. ($\trace XY = \svec(X)^T
\svec(Y)$, i.e. isometry)
$\sMat$ is inverse

Solve for $\svec(l)$:
\[
  L (\svec (l)) =  \svec (RHS).
\]


\end{slide}
\begin{slide}{}

$ L_{pq}= $
\[
  \left\{  \begin{array}{ll}
     2 \mu H^{(2)}_{F_{p_2},F_{p_1}}
  \left(
      \Lambda^{-1}_{F_{p_2},F_{q_1}} 
       \Lambda^{-1}_{F_{q_2},F_{p_1}}  +
      \Lambda^{-1}_{F_{p_1},F_{q_1}} 
       \Lambda^{-1}_{F_{q_2},F_{p_2}}  \right) \\
   ~~~~~~~~~~~~~~~~~~~~ \mbox{if } p \neq q, ~  k \neq l, ~ i \neq j;\\
     2\sqrt{2} \mu H^{(2)}_{F_{p_2},F_{p_1}}
  \left(\Lambda^{-1}_{F_{p_2},F_{q_2}} 
       \Lambda^{-1}_{F_{q_2},F_{p_1}}  \right) \\
   ~~~~~~~~~~~~~~~~~~~~~~ \mbox{if } p \neq q, ~  k \neq l, ~ i = j;\\
     2\sqrt{2} \mu H^{(2)}_{F_{p_2},F_{p_1}}\left(\Lambda^{-1}_{F_{p_2},F_{q_2}} 
       \Lambda^{-1}_{F_{q_2},F_{p_1}}  \right) \\
   ~~~~~~~~~~~~~~~~~~~~~~ \mbox{if } p = q, ~ k \neq l, ~  i = j;\\
     2 \mu H^{(2)}_{F_{p_2},F_{p_1}}
  \left(
      \Lambda^{-1}_{F_{p_2},F_{q_1}} 
       \Lambda^{-1}_{F_{q_2},F_{p_1}}  +
      \Lambda^{-1}_{F_{p_1},F_{q_1}} 
       \Lambda^{-1}_{F_{q_2},F_{p_2}}  \right) +1\\
   ~~~~~~~~~~~~~~~~~~~~~~~~~ \mbox{if } p = q, k \neq l, ~  i \neq j;
           \end{array}   \right.
\]

\end{slide}
\begin{slide}{}


\[
  \left\{  \begin{array}{ll}
     \sqrt{2}\mu \Lambda^{-1}_{F_{p_1},F_{q_1}} \Lambda^{-1}_{F_{q_2},F_{p_1}} \\
   ~~~~~~~~~~~~~~~~~~~~~~~~~~~ \mbox{if }   k=l, ~ i \neq j;\\
     \mu \Lambda^{-1}_{F_{p_1},F_{q_1}} \Lambda^{-1}_{F_{q_1},F_{p_1}}
   & \mbox{if }   k=l, ~ i=j\\
     2\sqrt(2) H^{(2)}_{F_{p_2},F_{p_1}}
   & \mbox{if }   w=1,~k \neq l\\
  1
   & \mbox{if }   w=1,~k = l.
\end{array}  \right.
\]

The $p$-th row
calculated using Hadamard product of pairs of columns of
$\Lambda^{-1}$,
\[
  \Lambda^{-1}_{F_{p_2},F_{:,1}} \circ
       \Lambda^{-1}_{F_{p_1},F_{:,2}}.
\]
complete vectorization



\end{slide}
\begin{slide}{}

$p=kl, k\leq l$, and last row, 
component of the right-hand-side of the system is\\
$RHS_p =$
\[
 \left\{
\begin{array}{lcl}
\sqrt{2}\left(  2H^{(2)}_{p} \circ
     \left\{\mu \Lambda^{-1}_p +A_p  
     -\alpha \right\} -\Lambda_p \right), 
                   & \mbox{if }k \neq l\\
      \mu \Lambda^{-1}_{kk} - \alpha & \mbox{if } k=l\\
     -\trace (\Lambda E)   & \mbox{last row } \\
\end{array}  \right.
\]

\end{slide}
\begin{slide}{}


\subsubsection{On a New Characterization of EDMs}
\label{sect:newedm}

The above application for EDM uses 
\[
D=\lambda E - P, ~ P \succeq 0, ~\diag P = \lambda e, ~\lambda \geq 0;
\]
this raises the {\bf question}:\\
\begin{quote}
{\em {\bf which}
Euclidean distance matrices $D$ can be expressed
as $D= \lambda (E - C)$ for some nonnegative scalar $\lambda$ and some 
correlation matrix $C$, where $E$ is the matrix of all ones.}
\end{quote}


We {\bf show (and characterize)} that the cones
\[  
\cone \left(E-\EE_n \right) \,  \varsubsetneq
\, \overline{\cone \left(E-\EE_n \right)} =\DD_n,
\]
where $\EE_n$ is the elliptope (set of correlation matrices) and $\DD_n$
is the (closed convex) cone of Euclidean distance matrices.

\end{slide}
\begin{slide}{}


The characterization is given using the 
Gale transform of the points generating $D$.
We also show that given points $p^1$, $p^2$, \ldots, $p^n \in \RR^r$,  
for any scalars $\lambda_1$, $\lambda_2$, \ldots,
$\lambda_n$ such that  
\[\sum_{j=1}^n \lambda_j \; p^j = 0, \;\;\;\;\;\;\;  \sum_{j=1}^n \lambda_j = 0,
\]       
we have  
\[ \sum_{j=1}^n \lambda_j \; \| p^i - p^j \|^2 = \alpha \mbox{ for all } i=1,\ldots,n, \]        
for some scalar $\alpha$ independent of $i$. 


\end{slide}
\begin{slide}{}


{\bf Recall}:\\
\begin{itemize}
\item
$n \times n$ matrix $D=(d_{ij})$ is 
{\em Euclidean distance matrix (EDM)} if 
\[
\exists \quad p^1,p^2,\ldots,p^n \in \RR^r, \mbox{ such that }
 \|p^i-p^j\|^2= d_{ij},\quad  \forall i,j
\]
\item
dimension of smallest Euclidean space containing $p^1,p^2, \ldots,p^n$ is 
the {\em embedding dimension} of $D$.
\item
$D$ (hollow) EDM $\iff$ $D$ negative semidefinite on $e^{\perp}$
\item
the set of $n \times n$ EDM matrices 
is a closed convex cone, $\DD_n$
\end{itemize}



\end{slide}
\begin{slide}{}


Since ${\EE}_n$ denotes set of $n \times n$ {\em correlation matrices}, 
i.e., set of positive semidefinite symmetric matrices
whose diagonal is equal to $e$, 
we are using the
well known fact \cite{MR98g:52001} 
that $\DD_n$ is the tangent cone of ${\EE}_n$ at $E$,
the matrix of all ones, i.e.,
\beq \label{eq:cones}    \DD_n = 
\overline{\cone \left(  E - \EE_n \right)} = 
\overline{\left\{ \lambda ( E - C ): \lambda \geq 0, C \in \EE_n \right\} },
\eeq
where $\bar{\cdot}$ denotes closure.

In general, it is difficult to determine whether the generated cone of a
set, $cone (C)$ is closed, though there are sufficient conditions, e.g.
$C$ is convex and compact and 0 is not in $C$.




\end{slide}
\begin{slide}{}

\begin{itemize}
\item
Let $V$ be $n \times (n-1)$  such that
\beq \label{Vmp} V^Te=0\;, \;\;\;\;  V^TV=I_{n-1} \;. \eeq  
\item
orthogonal projection on $M=e^{\perp}$, denoted by $J$, is 
$J:= VV^T = I - e e^T/n$. 
\item
$D$ hollow is EDM $\iff$
$ B := - \frac{1}{2} \; J D J \succeq 0$
\item
embedding dimension of $D$ is equal, $r$, rank of $B$. 
\item
the points $p^1, p^2, \ldots, p^n$ that 
generate $D$ are given by the rows of the $n \times r$ matrix
$P$ where $B:= P P^T$. 
\item
since $B e = 0$, centroid of the points $p^i$, $i=1,\ldots, n$
coincides with the origin
\end {itemize}



\end{slide}
\begin{slide}{}

Let $p^1,p^2,\ldots,p^n$ be points in $\RR^{r}$  whose 
centroid coincides with the origin, and 
are not contained in a proper hyperplane. Then   
\[ 
P := \left[ \begin{array}{c} {p^1}^T \\ {p^2}^T \\ \vdots \\ {p^n}^T 
  \end{array} \right]  
\] 
is of rank $r$. Let $B = P P^T$. 
Then it follows that the EDM matrix $D$ generated by 
$p^i$, $i=1,\ldots,n$ is given by
\beq \label{defK} D = 
\diag B  e^T + e  \left(\diag B\right)^T - 2 B.  \eeq  



\end{slide}
\begin{slide}{}

\begin{itemize}
\item
Let $\bar{r}=n-1-r$ and
\[ 
Z \mbox{ be } n \times \bar{r},  \mbox{ full column rank with}, ~
 {P}^T Z = 0,  e^T Z = 0
\]
\item
${z^i}^T$ denotes the $i$-th row of $Z$. i.e., 
\[ 
Z^T := \left[ {z^1} ~|~ {z^2} ~|~  \ldots ~|~  {z^n} \right].  
\] 
\item
$z^i$ is called Gale transform of $p^i$;
$Z$ is called a {\em Gale matrix} corresponding to $D$.
\item
The columns of $Z$ represent the {\em affine dependence relations} among the
points ${p}^1, {p}^2, \ldots , {p}^n$, 
i.e., among the rows of $P$.  
\end{itemize}

\end{slide}
\begin{slide}{}
{\bf Main Results}
\begin{thm} \label{thm1} 
Let $D$ be a Euclidean distance matrix and let $Z$ be a Gale matrix corresponding to $D$. Then,
the columns of $DZ$ are proportional to $e$.
\end{thm}
Equivalently
\begin{thm} 
Let $\lambda_1$, $\lambda_2$, \ldots, $\lambda_n$ be 
coefficients, not all zero, of the affine
dependence equation of the points ${p}^1, {p}^2, 
\ldots , {p}^n$, in $\RR^r$, i.e.,  
\[ \sum_{j=1}^n \lambda_j  p^j = 0, \quad \sum_{j=1}^n \lambda_j = 0.
    \]       
Then 
\[ \sum_{j=1}^n \lambda_j \; \| p^i - p^j \|^2 = \alpha \mbox{ for all } i=1,\ldots,n, \]        
for some scalar $\alpha$ independent of $i$. 
\end{thm}


\end{slide}
\begin{slide}{}

\begin{thm} \label{thm2} 
Let $D$ be a Euclidean distance matrix and let $Z$ be a Gale matrix 
corresponding to $D$. Then {\bf the following are equivalent}:
\begin{enumerate}
\item
\beq
D = \lambda ( E - C),
\end{equation}
for some nonnegative scalar $\lambda$ and some correlation matrix $C$; 
\item
\beq
 DZ=0.
\end{equation}
\end{enumerate}
\end{thm}




\end{slide}
\begin{slide}{}


{\bf Proof of the Main Results}\\

\begin{lem} \label{lem1}
Let $D$ be a Euclidean distance matrix and let 
$B= - \frac{1}{2} J D J$.  Then:
\begin{enumerate}
\item
\[ - \frac{1}{2} V^T D V = V^T B V;
\]
\item
\[
\NN(V^TDV)= \NN(P^T V).
\]
\end{enumerate}
\end{lem} 
\bpr
The first part follows directly from (\ref{defK}) and the 
definition of $V$. This yields the second
part since $B=PP^T$ and 
$\NN(V^T B V)= \NN(V^T P P^T V)= \NN(P^T V)$.   
\epr     


\end{slide}
\begin{slide}{}

\begin{lem} \label{lemLam}
Let $D$ be a Euclidean distance matrix and 
let $U$ be the matrix whose columns
form an orthonormal basis of the null space of $V^T D V$. Then $VU$ is a Gale matrix corresponding to $D$.  
\end{lem}

\bpr   
It follows from Lemma \ref{lem1} that $P^T V U = V^T D V U = 0$ and from the definition of $V$ in
(\ref{Vmp}) that  
$e^T V U =0$. Hence, the columns of     
$V U$ form an orthonormal basis
for the null space of  $\left[ \begin{array}{c}
           {P}^T  \\  e^T \end{array} \right] $.
\epr


\end{slide}
\begin{slide}{}

\noindent {\bf Proof of Theorem \ref{thm1}.}
Let $Z$ be a Gale matrix corresponding to $D$. Then  
It follows from Lemma \ref{lemLam} that 
$VU = Z Q$ for some nonsingular $\bar{r} \times \bar{r}$ matrix $Q$.  
Thus $V^T D Z = V^T D V U Q^{-1}= 0$. Hence, 
the columns of $D Z$ are proportional to $e$.
\epr 

\end{slide}
\begin{slide}{}


\noindent {\bf Proof of Theorem \ref{thm2}.}
$D = \lambda ( E - C)$ for some nonnegative scalar $\lambda$ and some correlation matrix $C$ 
if and only if $E - \frac{1}{\lambda} D$ is positive semidefinite.  
Let $Q= [ \frac{e}{\sqrt{n}} \;  V]$. Then, 
$E- D/\lambda \succeq 0 $  if and only if
$Q^T \; ( E - D/ \lambda ) \; Q \succeq 0 $. But 
\[ Q^T \; (E - D / \lambda )\; Q  =  \left[ \begin{array}{cc}
     n- \frac{1}{\lambda \; n } \; e^T D e & - \frac{1}{\lambda \; \sqrt{n}} \; e^T D V \\  
      - \frac{1}{\lambda \; \sqrt{n}} \; V^T D e  & - \frac{1}{\lambda } \; V^T D V  \\  
                                     \end{array} \right] .
\]
Recall that $ V^T (- D) V \succeq 0$ follows
from Lemma  \ref{lem1}.
Let $W$ and $U$ be the matrices whose columns form an orthonormal
basis for the range space and null space of $V^T (- D) V$, respectively. 


\end{slide}
\begin{slide}{}


Hence, $V^T (-D) V= W \Lambda W^T$,
where $\Lambda$ is the diagonal matrix of the positive eigenvalues of $V^T (-D) V$. Let  $Q^\prime =  
 \left[ \begin{array}{ccc}
           1  &  0 & 0 \\  0 & W  & U \end{array} \right]$.    
Then, $E- D/\lambda$ is positive semidefinite if and only if 
\beq \label{big} 
\begin{array}{rcl}
R &=&
{Q^\prime}^T  Q^T ( E - D/ \lambda) Q Q^\prime  \\
~\\
&=& 
\left[ \begin{array}{ccc}
   n- \frac{1}{\lambda  n }  e^T D e & - \frac{1}{\lambda  \sqrt{n}}  e^T D V W   
                          & - \frac{1}{\lambda \sqrt{n}}  e^T D V U  \\   
  - \frac{1}{\lambda \sqrt{n}} W^T V^T D e & \frac{1}{\lambda } \Lambda  &  0 \\  
  - \frac{1}{\lambda \sqrt{n}} U^T V^T D e  & 0  &  0  
                          \end{array} \right] \succeq 0.  
\end{array}
\eeq

\end{slide}
\begin{slide}{}

Now for sufficiently large  $\lambda$ the submatrix 
\[
 \left[ \begin{array}{cc}
     n- \frac{1}{\lambda \; n } \; e^T D e & - \frac{1}{\lambda \; \sqrt{n}} \; e^T D V W \\   
      - \frac{1}{\lambda \; \sqrt{n}} \; W^T V^T D e  &  \frac{1}{\lambda } \; \Lambda   \\  
                                  \end{array} \right] 
\]  
is positive definite. 

Thus $E - D/ \lambda$ is positive semidefinite 
if and only if
 $e^T D V U = e^T D Z= 0$. But it follows from Theorem \ref{thm1} that
$e^T D Z = 0$ if and only if
 $DZ=0$ and the result follows.     
\epr


\end{slide}
\begin{slide}{}

{\bf Given the two Euclidean distance matrices:}
\[ 
D_1 = \left[ \begin{array}{ccc} 0 & 1 & 4 \\ 1 & 0  & 1 \\  4 & 1 & 0 \end{array} \right], \;\;\;\;\;\;  
D_2 = \left[ \begin{array}{ccc} 0 & 1 & 0 \\ 1 & 0  & 1 \\  0 & 1 & 0 \end{array} \right],
\]
Gale matrices $Z_1,Z_2$ corresponding to $D_1,D_2$, resp. are:
\[ 
Z_1^T = \left[  1  ~ -2 ~  1  \right], \quad
Z_2^T = \left[  1  ~  0  ~ -1 \right],  
\]
Now $D_1 Z_1 = 2 e$ and $D_2 Z_2 = 0$.
 Then: $D_2 = E - C_2$, where   
\[ C_2 = \left[ \begin{array}{ccc} 1 & 0 & 1 \\ 0 & 1  & 0 \\  1 & 0 & 1 \end{array} \right] \succeq 0. \]
However, there exists no $\lambda \geq 0$ such that $D_1= \lambda ( E - C_1)$ for some correlation
matrix $C_1$.    


\label{endsect:newedm}
\label{endsect:matrixcompl}

\end{slide}
\begin{slide}{}

\appendix
\bs{Related Papers}
\subsection{Outline}
%Add nocite and others here and discuss references here.
Following are summaries of several papers closely related to the above
notes.
\begin{enumerate}
\item
Alizadeh has notes that get updated each time he gives the
course \cite{Alizadehnotes:00}.
\item
Nemirovski has course notes that he gave at Delft
\cite{bentalnemirov:98}.
\item
Todd, December, 2000 has a survey paper on SDP \cite{Todd:00}.

\end{enumerate}

\end{slide}
\begin{slide}{}
\bibliography{.master,.psd,.publs,.edm,.qap}
\end{slide}
%\end{multicols}
\end{document}
